Files
domainCheck/app/core/detect_engine.py
2026-04-14 22:53:52 +08:00

281 lines
11 KiB
Python

# -*- coding: UTF-8 -*-
'''
@Project :domainScanDemo
@File :detect_engine.py
@IDE :PyCharm
@Author :梦伴
@Date :2026/4/8 23:52
@explain : 检测引擎
'''
import time
from loguru import logger
from app.utils.database import Database
from app.detectors.rdap_detector import RDAPDetector
from app.detectors.wayback_detector import WaybackDetector
from app.detectors.baidu_detector import BaiduDetector
from app.detectors.qihu360_detector import Qihu360Detector
from app.detectors.google_detector import GoogleDetector
from app.detectors.chinaz_detector import ChinazDetector
from app.detectors.aizhan_detector import AizhanDetector
from app.detectors.juziseo_detector import JuziseoDetector
from app.detectors.jucha_detector import JuchaDetector
class DetectEngine:
"""
检测引擎
"""
def __init__(self):
"""
初始化检测引擎
"""
self.db = Database()
self.rdap_detector = RDAPDetector()
self.wayback_detector = WaybackDetector()
self.baidu_detector = BaiduDetector()
self.qihu360_detector = Qihu360Detector()
self.google_detector = GoogleDetector()
self.chinaz_detector = ChinazDetector()
self.aizhan_detector = AizhanDetector()
self.juziseo_detector = JuziseoDetector()
self.jucha_detector = JuchaDetector()
def detect_domain(self, domain_id):
"""
检测域名
:param domain_id: 域名ID
:return: bool - 是否检测成功
"""
try:
# 获取域名信息
domain_info = self.db.get_domain_by_id(domain_id)
if not domain_info:
logger.error(f"域名不存在: {domain_id}")
return False
domain = domain_info['domain']
logger.info(f"开始检测域名: {domain}")
# 更新检测状态为检测中
self.db.update_domain_detect_status(domain_id, 2) # 2 表示检测中
# 1. 基础检测
if not self._basic_detect(domain_id, domain):
logger.info(f"基础检测失败,停止后续检测: {domain}")
return False
# 2. 深度检测
if not self._deep_detect(domain_id, domain):
logger.info(f"深度检测失败: {domain}")
return False
# 更新检测状态为正常
self.db.update_domain_detect_status(domain_id, 3) # 3 表示正常
logger.info(f"域名检测完成: {domain}")
return True
except Exception as e:
logger.error(f"检测域名出错: {e}")
# 更新检测状态为检测失败
self.db.update_domain_detect_status(domain_id, 5) # 5 表示检测失败
return False
def _basic_detect(self, domain_id, domain):
"""
基础检测
:param domain_id: 域名ID
:param domain: 域名
:return: bool - 是否检测通过
"""
# 1. 检查是否为一口价域名
is_ykj = self.db.is_ykj_domain(domain_id)
# 2. 注册状态检测(一口价域名跳过)
if not is_ykj:
register_status = self.rdap_detector.check_register_status(domain)
self.db.update_domain_register_status(domain_id, register_status)
# 3. 黑名单缓存检查
if self.db.is_blacklisted(domain):
logger.info(f"域名在黑名单中: {domain}")
self.db.update_domain_detect_status(domain_id, 4) # 4 表示黑名单
return False
# 4. 时光机快照年份采集
snapshot_years = self.wayback_detector.get_snapshot_years(domain)
if snapshot_years:
self.db.update_domain_snapshot_years(domain_id, ','.join(map(str, snapshot_years)))
# 5. 时光机正文抽样与敏感词匹配
if self.wayback_detector.has_sensitive_content(domain):
logger.info(f"域名包含敏感词: {domain}")
self.db.update_domain_detect_status(domain_id, 4) # 4 表示黑名单
self.db.add_to_blacklist(domain, "快照包含敏感词")
return False
return True
def _deep_detect(self, domain_id, domain):
"""
深度检测
:param domain_id: 域名ID
:param domain: 域名
:return: bool - 是否检测通过
"""
# 1. 百度历史/Site
baidu_history = self.baidu_detector.check_history(domain)
baidu_site = self.baidu_detector.check_site(domain)
# 2. 360 Site
qihu360_site = self.qihu360_detector.check_site(domain)
# 3. Google Site
google_site = self.google_detector.check_site(domain)
# 4. 站长之家
chinaz_info = self.chinaz_detector.check_domain(domain)
# 5. 爱站网
aizhan_info = self.aizhan_detector.check_domain(domain)
# 6. 桔子SEO
juziseo_info = self.juziseo_detector.check_domain(domain)
# 7. 聚查
jucha_info = self.jucha_detector.check_domain(domain)
# 检查是否有风险
if self._check_risk(domain_id, domain, baidu_history, baidu_site, qihu360_site, google_site, chinaz_info, aizhan_info, juziseo_info, jucha_info):
return False
# 保存检测结果
self.db.add_detection_result(domain_id, baidu_history, baidu_site, qihu360_site, google_site, chinaz_info, aizhan_info, juziseo_info, jucha_info)
return True
def _check_risk(self, domain_id, domain, baidu_history, baidu_site, qihu360_site, google_site, chinaz_info, aizhan_info, juziseo_info, jucha_info):
"""
检查风险
:param domain_id: 域名ID
:param domain: 域名
:param baidu_history: 百度历史
:param baidu_site: 百度site
:param qihu360_site: 360 site
:param google_site: Google site
:param chinaz_info: 站长之家信息
:param aizhan_info: 爱站网信息
:param juziseo_info: 桔子SEO信息
:param jucha_info: 聚查信息
:return: bool - 是否有风险
"""
# 检查百度历史过灰
if baidu_history and '' in str(baidu_history):
logger.info(f"百度历史过灰: {domain}")
self.db.update_domain_detect_status(domain_id, 4) # 4 表示黑名单
self.db.add_to_blacklist(domain, "百度历史过灰")
return True
# 检查标题敏感词
if chinaz_info and 'title' in chinaz_info:
if self._contains_sensitive_words(chinaz_info['title']):
logger.info(f"标题包含敏感词: {domain}")
self.db.update_domain_detect_status(domain_id, 4) # 4 表示黑名单
self.db.add_to_blacklist(domain, "标题包含敏感词")
return True
# 检查子域名
if baidu_site and 'subdomains' in baidu_site:
subdomains = baidu_site['subdomains']
# 排除 www, @, m
valid_subdomains = [sub for sub in subdomains if sub not in ['www', '@', 'm']]
if valid_subdomains:
logger.info(f"存在子域名: {domain}")
self.db.update_domain_detect_status(domain_id, 4) # 4 表示黑名单
self.db.add_to_blacklist(domain, "存在子域名")
return True
# 检查风险提示
if aizhan_info and 'risk' in aizhan_info:
if aizhan_info['risk'] in ['低风险', '疑似色情博彩风险', '严重影响权重']:
logger.info(f"风险提示: {domain}")
self.db.update_domain_detect_status(domain_id, 4) # 4 表示黑名单
self.db.add_to_blacklist(domain, f"风险提示: {aizhan_info['risk']}")
return True
# 检查WHOIS状态
if jucha_info and 'whois' in jucha_info:
if jucha_info['whois'].get('status') in ['clientHold', 'serverHold']:
logger.info(f"WHOIS状态异常: {domain}")
self.db.update_domain_detect_status(domain_id, 4) # 4 表示黑名单
self.db.add_to_blacklist(domain, "WHOIS状态异常")
return True
# 检查拦截检测
if jucha_info and 'intercept' in jucha_info:
if not jucha_info['intercept'].get('normal', True):
logger.info(f"拦截检测异常: {domain}")
self.db.update_domain_detect_status(domain_id, 4) # 4 表示黑名单
self.db.add_to_blacklist(domain, "拦截检测异常")
return True
return False
def _contains_sensitive_words(self, text):
"""
检查文本是否包含敏感词
:param text: 文本
:return: bool - 是否包含敏感词
"""
# 这里可以添加敏感词检查逻辑
sensitive_words = ['色情', '博彩', '赌博', '毒品', '暴力']
for word in sensitive_words:
if word in text:
return True
return False
def process_task(self, task_id):
"""
处理检测任务
:param task_id: 任务ID
:return: bool - 是否处理成功
"""
try:
# 获取任务信息
task = self.db.get_task_by_id(task_id)
if not task:
logger.error(f"任务不存在: {task_id}")
return False
domain_id = task['domain_id']
# 更新任务状态为执行中
self.db.update_task_status(task_id, 1) # 1 表示执行中
# 执行检测
success = self.detect_domain(domain_id)
# 更新任务状态
if success:
self.db.update_task_status(task_id, 2) # 2 表示完成
else:
# 增加重试次数
retry_count = task.get('retry_count', 0) + 1
if retry_count < 3:
self.db.update_task_retry_count(task_id, retry_count)
self.db.update_task_status(task_id, 0) # 0 表示待执行
else:
self.db.update_task_status(task_id, 3) # 3 表示失败
return success
except Exception as e:
logger.error(f"处理任务出错: {e}")
# 更新任务状态为失败
self.db.update_task_status(task_id, 3) # 3 表示失败
return False