[经验分享] 《JS跳转代码》根据爬虫、设备、来源判断,实现js屏蔽/跳转与展示控制

[复制链接]
sheep 发表于 2022-12-24 16:15 | 显示全部楼层 |阅读模式
JS 多场景跳转模板,内置完整蜘蛛 UA 库,支持三种常用访问控制,配置区域配置对应模块即可,自带防循环重定向,可跳转域名或目录。

《爬虫/蜘蛛/设备标识全量识别清单》


本文中的爬虫/蜘蛛/设备标识更新/补充,均会在上述文章内同步,复制代码后注意核对

  1. <script type="text/javascript">
  2. // ==============================================
  3. // JS 跳转控制通用模板
  4. // 功能:爬虫识别 | 设备识别 | 来源识别 | 防循环重定向
  5. // 支持:域名前缀跳转 / 完整域名跳转 / 指定目录跳转
  6. // ==============================================

  7. // ====================== 【配置区】 按需修改即可 ======================
  8. // 1. 通用配置
  9. const redirect_mode = 'replace';       // 跳转方式:replace=无历史记录跳转  href=保留历史记录跳转
  10. //replace:覆盖当前历史,返回按钮回不到跳转前页面
  11. //href:新增历史,返回可以回到跳转前页面

  12. // 2. 爬虫配置
  13. const spider_redirect = '';            // 爬虫跳转目标,空=放行不跳转;示例:/spider.html 或 https://xxx.com

  14. // 3. 设备跳转配置(留空=该设备不执行跳转)
  15. const pc_redirect_target = '';         // PC端跳转目标;示例:www1.${base_domain} 或 /pc/
  16. const tablet_redirect_target = '';     // 平板端(iPad等)跳转目标;示例:www1.${base_domain} 或 /pad/
  17. const mobile_redirect_target = '';     // 手机端跳转目标;示例:m.${base_domain} 或 /mobile/

  18. // 4. 来源跳转配置
  19. const non_search_redirect = '';        // 非搜索引擎&非AI来源跳转目标,空=不启用;示例:/error.html
  20. const ai_referer_redirect = '';        // AI平台来源单独跳转目标,空=不单独处理

  21. // 5. 防循环重定向规则(匹配该规则则取消跳转,避免死循环)
  22. // 域名前缀跳转填:^www1\. ;目录跳转填:^/mobile/
  23. const loop_guard_pattern = '';
  24. // ====================== 【配置区结束】 ======================


  25. // ---------------------- 1. 基础域名提取 ----------------------
  26. // 自动移除 www. / www1. / m. / wap. 前缀,提取裸域名
  27. let base_domain = location.host;
  28. const hostMatch = location.host.match(/^(www|www1|m|wap)\.(.+)$/i);
  29. if (hostMatch) {
  30.     base_domain = hostMatch[2];
  31. }


  32. // ---------------------- 2. 设备类型精准识别 ----------------------
  33. // 初始化标记
  34. let is_pc = false;
  35. let is_tablet = false;
  36. let is_mobile = false;
  37. const ua = navigator.userAgent || '';

  38. // 优先级1:识别手机端
  39. if (/mobile|iphone|ipod|android.*mobile|windows\s*phone|blackberry|opera\s*mini|ucweb|meego|silk|kindle|playbook|nintendo|playstation|wii/i.test(ua)) {
  40.     is_mobile = true;
  41. }

  42. // 优先级2:识别平板端(iPad/安卓平板,避免误判为PC)
  43. if (!is_mobile) {
  44.     if (/ipad|android.*tablet|tablet/i.test(ua)) {
  45.         is_tablet = true;
  46.     }
  47. }

  48. // 优先级3:剩余识别为PC桌面端
  49. if (!is_mobile && !is_tablet) {
  50.     if (/windows\s*nt|macintosh|mac\s*os\s*x|x11|cros|linux\s*(x86|amd64|i686)/i.test(ua)) {
  51.         is_pc = true;
  52.     }
  53. }


  54. // ---------------------- 3. 爬虫/蜘蛛全量识别 ----------------------
  55. // 覆盖范围:通用泛匹配 + 全球主流AI蜘蛛 + 中国主流AI蜘蛛 + 全球搜索引擎 + 中国搜索引擎 + 其他平台蜘蛛
  56. let is_spider = false;
  57. const spiderRegex = /bot|spider|crawl|crawler|index|archive|gptbot|oai-searchbot|chatgpt-user|claudebot|claude-searchbot|claude-user|xai-searchbot|xai-bot|xai-grok|meta-externalagent|meta-externalfetcher|perplexitybot|perplexity-user|huggingfacebot|amazonbot|coherebot|inflectionbot|google-extended|ccbot|doubaobot|yuanbaobot|hunyuanbot|qwenbot|tongyibot|wenxinbot|moonshotbot|deepseekbot|kimibot|zhipubot|youbot|google|bing|yahoo|yandex|duckduckbot|slurp|applebot|baidu|sogou|bytespider|shenma|youdao|360spider|soso|toutiao|semrush|ahrefs|petalbot|exabot|facebot|facebook|twitter|linkedin|telegram|ia_archiver/i;
  58. if (spiderRegex.test(ua)) {
  59.     is_spider = true;
  60. }


  61. // ---------------------- 4. 访问来源(Referrer)识别 ----------------------
  62. let is_search_referer = false;
  63. let is_ai_referer = false;
  64. const referrer = document.referrer || '';

  65. // 识别搜索引擎来源(支持多级域名,如google.co.jp、baidu.com.cn)
  66. const searchReg = /\.(google|bing|yahoo|yandex|duckduckgo|apple|baidu|sogou|shenma|sm|youdao|360|soso|toutiao)(\.[a-z0-9\-]+){0,2}\//i;
  67. if (searchReg.test(referrer)) {
  68.     is_search_referer = true;
  69. }

  70. // 识别AI平台来源
  71. const aiReg = /\.(openai|anthropic|claude|xai|perplexity|huggingface|cohere|inflection|meta|gemini|doubao|wenxin|tongyi|qianwen|hunyuan|moonshot|deepseek)(\.[a-z0-9\-]+){0,2}\//i;
  72. if (aiReg.test(referrer)) {
  73.     is_ai_referer = true;
  74. }


  75. // ---------------------- 5. 跳转目标计算(优先级从高到低) ----------------------
  76. let final_target = '';

  77. // 优先级1:爬虫专属逻辑
  78. if (is_spider) {
  79.     final_target = spider_redirect;
  80. }

  81. // 优先级2:AI平台来源单独跳转(非爬虫时生效)
  82. if (!is_spider && is_ai_referer) {
  83.     final_target = ai_referer_redirect;
  84. }

  85. // 优先级3:非搜索&非AI来源跳转(非爬虫、未触发AI跳转时生效)
  86. if (!is_spider && !is_search_referer && !is_ai_referer) {
  87.     final_target = non_search_redirect;
  88. }

  89. // 优先级4:设备维度跳转(非爬虫、未触发来源跳转时生效)
  90. if (!is_spider && final_target === '') {
  91.     if (is_pc) {
  92.         final_target = pc_redirect_target;
  93.     }
  94.     if (is_tablet) {
  95.         final_target = tablet_redirect_target;
  96.     }
  97.     if (is_mobile) {
  98.         final_target = mobile_redirect_target;
  99.     }
  100. }

  101. // 处理域名前缀模板变量,替换 ${base_domain} 为实际裸域名
  102. if (final_target && final_target.includes('${base_domain}')) {
  103.     final_target = final_target.replace('${base_domain}', base_domain);
  104. }


  105. // ---------------------- 6. 防循环重定向检测 ----------------------
  106. if (loop_guard_pattern !== '') {
  107.     const loopReg = new RegExp(loop_guard_pattern, 'i');
  108.     // 域名前缀场景防护
  109.     if (loopReg.test(location.host)) {
  110.         final_target = '';
  111.     }
  112.     // 目录跳转场景防护
  113.     if (loopReg.test(location.pathname)) {
  114.         final_target = '';
  115.     }
  116. }


  117. // ---------------------- 7. 执行跳转 ----------------------
  118. if (final_target !== '') {
  119.     // 目标包含http/https → 直接跳转到完整域名
  120.     if (/^https?:\/\//i.test(final_target)) {
  121.         redirect_mode === 'replace'
  122.             ? window.location.replace(final_target)
  123.             : window.location.href = final_target;
  124.     }
  125.     // 目标以/开头 → 站内目录跳转
  126.     else if (final_target.startsWith('/')) {
  127.         redirect_mode === 'replace'
  128.             ? window.location.replace(final_target)
  129.             : window.location.href = final_target;
  130.     }
  131.     // 其他情况 → 视为域名前缀,拼接完整域名+原请求路径
  132.     else {
  133.         const fullUrl = 'http://' + final_target + location.pathname + location.search;
  134.         redirect_mode === 'replace'
  135.             ? window.location.replace(fullUrl)
  136.             : window.location.href = fullUrl;
  137.     }
  138. }
  139. </script>
复制代码


下一页是独立功能版本
您需要登录后才可以回帖 登录 | 立即注册

本版积分规则

关注公众号
Archiver|手机版|小黑屋|社区规范|绵羊优创

相关侵权、举报、投诉及建议等,请发 E-mail:2363400792@qq.com

Powered by Discuz! X5.0 © 2001-2026 Discuz! Team.|京ICP备19037745号-2公安备案京公网安备11011502037529号

在本版发帖
关注公众号
QQ客服返回顶部
来点音乐
优聚封面
歌曲名称
歌手名称
0:00 0:00
顺序播放
歌词加载中...