《PHP跳转代码》根据爬虫、设备、来源判断,实现php屏蔽/跳转与展示控制
PHP多场景跳转模板,内置完整蜘蛛 UA 库,支持三种常用访问控制,配置区域配置对应模块即可,自带防循环重定向,可跳转域名或目录。《爬虫/蜘蛛/设备标识全量识别清单》
http://www.sheepyc.com/thread-703-1-1.html
本文中的爬虫/蜘蛛/设备标识更新/补充,均会在上述文章内同步,复制代码后注意核对
<?php
// ==============================================
// PHP 跳转控制通用模板
// 功能:爬虫识别 | 设备识别 | 来源识别 | 防循环重定向
// 支持:域名前缀跳转 / 完整域名跳转 / 指定目录跳转
// ==============================================
// ====================== 【配置区】 按需修改即可 ======================
// 1. 通用配置
$redirect_code = 302; // 重定向状态码:301=永久重定向302=临时重定向
// 2. 爬虫配置
$spider_redirect = ''; // 爬虫跳转目标,空=放行不跳转;示例:/spider.html 或 https://xxx.com
// 3. 设备跳转配置(留空=该设备不执行跳转)
$pc_redirect_target = ''; // PC端跳转目标;示例:www1.$base_domain 或 /pc/
$tablet_redirect_target = '';// 平板端(iPad等)跳转目标;示例:www1.$base_domain 或 /pad/
$mobile_redirect_target = '';// 手机端跳转目标;示例:m.$base_domain 或 /mobile/
// 4. 来源跳转配置
$non_search_redirect = ''; // 非搜索引擎&非AI来源跳转目标,空=不启用;示例:/error.html
$ai_referer_redirect = ''; // AI平台来源单独跳转目标,空=不单独处理
// 5. 防循环重定向规则(匹配该规则则取消跳转,避免死循环)
// 域名前缀跳转填:^www1\. ;目录跳转填:^/mobile/
$loop_guard_pattern = '';
// ====================== 【配置区结束】 ======================
// ---------------------- 1. 基础域名提取 ----------------------
// 自动移除 www. / www1. / m. / wap. 前缀,提取裸域名
$base_domain = $_SERVER['HTTP_HOST'] ?? '';
if (preg_match('/^(www|www1|m|wap)\.(.+)$/i', $base_domain, $matches)) {
$base_domain = $matches;
}
// ---------------------- 2. 设备类型精准识别 ----------------------
// 初始化标记
$is_pc = false;
$is_tablet = false;
$is_mobile = false;
$ua = $_SERVER['HTTP_USER_AGENT'] ?? '';
// 优先级1:识别手机端
if (preg_match('/mobile|iphone|ipod|android.*mobile|windows\s*phone|blackberry|opera\s*mini|ucweb|meego|silk|kindle|playbook|nintendo|playstation|wii/i', $ua)) {
$is_mobile = true;
}
// 优先级2:识别平板端(iPad/安卓平板,避免误判为PC)
if (!$is_mobile) {
if (preg_match('/ipad|android.*tablet|tablet/i', $ua)) {
$is_tablet = true;
}
}
// 优先级3:剩余识别为PC桌面端
if (!$is_mobile && !$is_tablet) {
if (preg_match('/windows\s*nt|macintosh|mac\s*os\s*x|x11|cros|linux\s*(x86|amd64|i686)/i', $ua)) {
$is_pc = true;
}
}
// ---------------------- 3. 爬虫/蜘蛛全量识别 ----------------------
// 覆盖范围:通用泛匹配 + 全球主流AI蜘蛛 + 中国主流AI蜘蛛 + 全球搜索引擎 + 中国搜索引擎 + 其他平台蜘蛛
$is_spider = false;
$spiderRegex = '/bot|spider|crawl|crawler|index|archive|gptbot|oai-searchbot|chatgpt-user|claudebot|claude-searchbot|claude-user|xai-searchbot|xai-bot|xai-grok|meta-externalagent|meta-externalfetcher|perplexitybot|perplexity-user|huggingfacebot|amazonbot|coherebot|inflectionbot|google-extended|ccbot|doubaobot|yuanbaobot|hunyuanbot|qwenbot|tongyibot|wenxinbot|moonshotbot|deepseekbot|kimibot|zhipubot|youbot|google|bing|yahoo|yandex|duckduckbot|slurp|applebot|baidu|sogou|bytespider|shenma|youdao|360spider|soso|toutiao|semrush|ahrefs|petalbot|exabot|facebot|facebook|twitter|linkedin|telegram|ia_archiver/i';
if (preg_match($spiderRegex, $ua)) {
$is_spider = true;
}
// ---------------------- 4. 访问来源(Referrer)识别 ----------------------
$is_search_referer = false;
$is_ai_referer = false;
$referrer = $_SERVER['HTTP_REFERER'] ?? '';
// 识别搜索引擎来源(支持多级域名,如google.co.jp、baidu.com.cn)
$searchReg = '/\.(google|bing|yahoo|yandex|duckduckgo|apple|baidu|sogou|shenma|sm|youdao|360|soso|toutiao)(\.+){0,2}\//i';
if (preg_match($searchReg, $referrer)) {
$is_search_referer = true;
}
// 识别AI平台来源
$aiReg = '/\.(openai|anthropic|claude|xai|perplexity|huggingface|cohere|inflection|meta|gemini|doubao|wenxin|tongyi|qianwen|hunyuan|moonshot|deepseek)(\.+){0,2}\//i';
if (preg_match($aiReg, $referrer)) {
$is_ai_referer = true;
}
// ---------------------- 5. 跳转目标计算(优先级从高到低) ----------------------
$final_target = '';
// 优先级1:爬虫专属逻辑
if ($is_spider) {
$final_target = $spider_redirect;
}
// 优先级2:AI平台来源单独跳转(非爬虫时生效)
if (!$is_spider && $is_ai_referer) {
$final_target = $ai_referer_redirect;
}
// 优先级3:非搜索&非AI来源跳转(非爬虫、未触发AI跳转时生效)
if (!$is_spider && !$is_search_referer && !$is_ai_referer) {
$final_target = $non_search_redirect;
}
// 优先级4:设备维度跳转(非爬虫、未触发来源跳转时生效)
if (!$is_spider && $final_target === '') {
if ($is_pc) {
$final_target = $pc_redirect_target;
}
if ($is_tablet) {
$final_target = $tablet_redirect_target;
}
if ($is_mobile) {
$final_target = $mobile_redirect_target;
}
}
// 处理域名前缀变量,替换 $base_domain 为实际裸域名
if ($final_target !== '' && strpos($final_target, '$base_domain') !== false) {
$final_target = str_replace('$base_domain', $base_domain, $final_target);
}
// ---------------------- 6. 防循环重定向检测 ----------------------
if ($loop_guard_pattern !== '') {
// 域名前缀场景防护
if (preg_match('/' . $loop_guard_pattern . '/i', $_SERVER['HTTP_HOST'])) {
$final_target = '';
}
// 目录跳转场景防护
if (preg_match('/' . $loop_guard_pattern . '/i', $_SERVER['REQUEST_URI'])) {
$final_target = '';
}
}
// ---------------------- 7. 执行跳转 ----------------------
if ($final_target !== '') {
// 目标包含http/https → 直接跳转到完整域名
if (preg_match('/^https?:\/\//i', $final_target)) {
header('Location: ' . $final_target, true, $redirect_code);
exit;
}
// 目标以/开头 → 站内目录跳转
else if (str_starts_with($final_target, '/')) {
header('Location: ' . $final_target, true, $redirect_code);
exit;
}
// 其他情况 → 视为域名前缀,拼接完整域名+原请求路径
else {
$fullUrl = 'http://' . $final_target . $_SERVER['REQUEST_URI'];
header('Location: ' . $fullUrl, true, $redirect_code);
exit;
}
}
?>
下一页是独立功能版本
独立功能版本:
版本 1:php 网站跳转代码(userAgent 爬虫判断,蜘蛛正常,用户任何形式进入网站都跳转)
(直接粘贴到页面最顶部,任何 HTML 输出、空格、换行、BOM 头之前加载)
<?php
// 爬虫UA特征正则 - 匹配优先级按分类从高到低排列
// 分类顺序:通用泛匹配 > 全球主流AI蜘蛛 > 中国主流AI蜘蛛 > 全球主流搜索引擎 > 中国主流搜索引擎 > 其他类爬虫
$spiderRegex = '/bot|spider|crawl|crawler|index|archive|gptbot|oai-searchbot|chatgpt-user|claudebot|claude-searchbot|claude-user|xai-searchbot|xai-bot|xai-grok|meta-externalagent|meta-externalfetcher|perplexitybot|perplexity-user|huggingfacebot|amazonbot|coherebot|inflectionbot|google-extended|ccbot|doubaobot|yuanbaobot|hunyuanbot|qwenbot|tongyibot|wenxinbot|moonshotbot|deepseekbot|kimibot|zhipubot|youbot|google|bing|yahoo|yandex|duckduckbot|slurp|applebot|baidu|sogou|bytespider|shenma|youdao|360spider|soso|semrush|ahrefs|petalbot|exabot|facebot|facebook|twitter|linkedin|telegram|ia_archiver/i';
// 兼容UA为空的边界情况
$userAgent = isset($_SERVER['HTTP_USER_AGENT']) ? $_SERVER['HTTP_USER_AGENT'] : '';
$isSpider = preg_match($spiderRegex, $userAgent);
if (!$isSpider) {
// 真实用户访问:跳转到指定页面
// 请修改下方路径为你的目标跳转地址
header('Location: /error.html');
exit; // 终止后续代码执行,必须添加
}
// 爬虫访问:正常渲染页面,不做跳转
// 可在此处补充爬虫专属处理逻辑(如输出专属内容、统计爬虫访问等)
?>
版本 2:php 网站跳转代码(referrer 来源页面判断,蜘蛛正常,用户通过搜索引擎 / AI 平台进入不跳转,其他任何形式进入全部跳转)
(直接粘贴到页面**最顶部**,任何 HTML 输出、空格、换行、BOM 头之前加载)
<?php
// 访问控制逻辑:官方蜘蛛正常访问 | 搜索引擎来源正常访问 | AI平台来源正常访问 | 其余访问跳转
// 蜘蛛UA分类顺序:通用泛匹配 > 全球主流AI蜘蛛 > 中国主流AI蜘蛛 > 全球主流搜索引擎 > 中国主流搜索引擎 > 其他类爬虫
$spiderRegex = '/bot|spider|crawl|crawler|index|archive|gptbot|oai-searchbot|chatgpt-user|claudebot|claude-searchbot|claude-user|xai-searchbot|xai-bot|xai-grok|meta-externalagent|meta-externalfetcher|perplexitybot|perplexity-user|huggingfacebot|amazonbot|coherebot|inflectionbot|google-extended|ccbot|doubaobot|yuanbaobot|hunyuanbot|qwenbot|tongyibot|wenxinbot|moonshotbot|deepseekbot|kimibot|zhipubot|youbot|google|bing|yahoo|yandex|duckduckbot|slurp|applebot|baidu|sogou|bytespider|shenma|youdao|360spider|soso|semrush|ahrefs|petalbot|exabot|facebot|facebook|twitter|linkedin|telegram|ia_archiver/i';
// 搜索引擎来源域名正则 - 匹配全球+中国主流搜索引擎referrer,支持多级域名
$searchReferrerRegex = '/\.(google|bing|yahoo|yandex|duckduckgo|apple|baidu|sogou|shenma|sm|youdao|360|soso|toutiao)(\.+){0,2}\//i';
// AI平台来源域名正则 - 匹配全球+中国主流AI平台referrer,支持多级域名
$aiReferrerRegex = '/\.(openai|anthropic|claude|xai|perplexity|huggingface|cohere|inflection|meta|gemini|doubao|wenxin|tongyi|qianwen|hunyuan|moonshot|deepseek)(\.+){0,2}\//i';
// 1. 优先判断是否为官方蜘蛛爬虫(蜘蛛不受来源限制,一律放行)
$userAgent = isset($_SERVER['HTTP_USER_AGENT']) ? $_SERVER['HTTP_USER_AGENT'] : '';
$isSpider = preg_match($spiderRegex, $userAgent);
if (!$isSpider) {
// 2. 普通用户则判断访问来源:搜索引擎 或 AI平台 均放行
$referrer = isset($_SERVER['HTTP_REFERER']) ? $_SERVER['HTTP_REFERER'] : '';
$isFromSearchEngine = preg_match($searchReferrerRegex, $referrer);
$isFromAIPlatform = preg_match($aiReferrerRegex, $referrer);
if (!$isFromSearchEngine && !$isFromAIPlatform) {
// 直接访问 / 非搜索引擎&非AI平台来源:跳转到指定页面
// 请修改下方路径为你的目标跳转地址
header('Location: /error.html');
exit; // 终止后续代码执行,必须添加
}
}
// 爬虫访问 / 搜索引擎来源 / AI平台来源访客:正常渲染页面
// 可在此处补充爬虫专属处理逻辑(如统计爬虫访问量)
?>
页:
[1]