diff --git a/app.py b/app.py index 0dbf048..22a5c22 100644 --- a/app.py +++ b/app.py @@ -1909,6 +1909,56 @@ def create_app(config_name='default'): 'message': f'抓取失败: {str(e)}' }), 500 + @app.route('/api/fetch-skill-info', methods=['POST']) + @login_required + def fetch_skill_info(): + """抓取 Skill 信息(GitHub 仓库)""" + # 只允许管理员访问 + if not isinstance(current_user, AdminModel): + return jsonify({'success': False, 'message': '无权访问'}), 403 + + try: + data = request.get_json() + url = data.get('url', '').strip() + + if not url: + return jsonify({ + 'success': False, + 'message': '请提供 GitHub 仓库URL' + }), 400 + + # 创建抓取器 + fetcher = WebsiteFetcher(timeout=15) + + # 抓取 Skill 信息 + info = fetcher.fetch_skill_info(url) + + if not info: + return jsonify({ + 'success': False, + 'message': '无法获取 Skill 信息,请检查 URL 是否为有效的 GitHub 仓库' + }) + + return jsonify({ + 'success': True, + 'data': { + 'name': info.get('name', ''), + 'short_desc': info.get('short_desc', ''), + 'description': info.get('description', ''), + 'github_url': info.get('github_url', ''), + 'source_repo': info.get('source_repo', ''), + 'source_type': 'github', + 'usage': info.get('usage', ''), + 'examples': info.get('examples', '') + } + }) + + except Exception as e: + return jsonify({ + 'success': False, + 'message': f'抓取失败: {str(e)}' + }), 500 + @app.route('/api/upload-logo', methods=['POST']) @login_required def upload_logo(): @@ -3417,6 +3467,7 @@ Sitemap: {}sitemap.xml # Skills管理视图 class SkillAdmin(SecureModelView): + create_template = 'admin/skill/create.html' can_edit = True can_delete = True can_create = True diff --git a/templates/admin/skill/create.html b/templates/admin/skill/create.html new file mode 100644 index 0000000..8f1f410 --- /dev/null +++ b/templates/admin/skill/create.html @@ -0,0 +1,224 @@ +{% extends 'admin/model/create.html' %} + +{% block tail %} +{{ super() }} + +{% endblock %} + +{% block header %} +

添加 Skill

+{% endblock %} + +{% block form %} +
+ +
+ + +
+
+
+ +
+ + +
+ +
+ + +
+ +
+ + +
+ +
+ + +
+ Logo预览 +
+
+ +
+ + +
+ +
+ + +
+ +
+ + +
+ +
+ + +
+ +
+ + +
+ +
+
+ + +
+
+ +
+
+ + +
+
+ +
+ + +
+ +
+ + 取消 +
+{% endblock %} + +{% block tail2 %} + +{% endblock %} diff --git a/utils/website_fetcher.py b/utils/website_fetcher.py index b3abe93..186bcbd 100644 --- a/utils/website_fetcher.py +++ b/utils/website_fetcher.py @@ -6,6 +6,7 @@ import requests from bs4 import BeautifulSoup from urllib.parse import urljoin, urlparse import os +import re from PIL import Image from io import BytesIO @@ -54,6 +55,243 @@ class WebsiteFetcher: print(f"抓取网站信息失败: {str(e)}") return None + def fetch_skill_info(self, url): + """ + 抓取 Skill 信息(GitHub 仓库) + + Args: + url: GitHub 仓库URL + + Returns: + dict: 包含name, short_desc, github_url, source_repo的字典 + """ + try: + # 解析 GitHub URL + # 支持格式: https://github.com/user/repo 或 https://github.com/user/repo/tree/main + parsed = urlparse(url) + if 'github.com' not in parsed.netloc: + return None + + path_parts = [p for p in parsed.path.split('/') if p] + if len(path_parts) < 2: + return None + + owner = path_parts[0] + repo = path_parts[1].replace('.git', '') + + # 获取仓库信息 + api_url = f"https://api.github.com/repos/{owner}/{repo}" + response = requests.get(api_url, headers=self.headers, timeout=self.timeout) + if response.status_code != 200: + return None + + repo_info = response.json() + + # 获取 .claude 目录下的 skills + skills_url = f"https://api.github.com/repos/{owner}/{repo}/contents/.claude/skills" + skills_response = requests.get(skills_url, headers=self.headers, timeout=self.timeout) + + skills = [] + if skills_response.status_code == 200: + skills_data = skills_response.json() + for item in skills_data: + if item.get('type') == 'file' and item['name'].endswith('.md'): + # 获取 skill 文件内容 + skill_content = requests.get(item['download_url'], headers=self.headers, timeout=self.timeout) + if skill_content.status_code == 200: + skill_info = self._parse_skill_file( + item['name'], + skill_content.text, + item.get('html_url') or item.get('download_url') or '' + ) + if skill_info: + skills.append(skill_info) + + summary = self._summarize_skills(repo_info, skills) + + return { + 'name': summary.get('name') or repo_info.get('name', ''), + 'short_desc': summary.get('short_desc') or repo_info.get('description', ''), + 'description': summary.get('description', ''), + 'github_url': url, + 'source_repo': f"{owner}/{repo}", + 'usage': summary.get('usage', ''), + 'examples': summary.get('examples', ''), + 'skills': skills + } + + except Exception as e: + print(f"抓取 Skill 信息失败: {str(e)}") + return None + + def _parse_skill_file(self, filename, content, source_url=''): + """解析 skill 文件内容""" + try: + lines = [line.rstrip() for line in content.strip().split('\n')] + fallback_name = filename.replace('.md', '') + title = fallback_name + + for line in lines: + if line.startswith('# '): + title = line[2:].strip() + break + + sections = self._extract_markdown_sections(lines) + + desc = self._pick_section( + sections, + ['描述', '简介', 'description', 'overview', 'summary', 'what it does'] + ) + use_cases = self._pick_section( + sections, + ['应用场景', '适用场景', '使用场景', 'use cases', 'when to use', 'use case'] + ) + usage = self._pick_section( + sections, + ['使用方法', '用法', 'usage', 'how to use', 'quick start'] + ) + examples = self._pick_section( + sections, + ['示例', 'examples', 'example'] + ) + + if not desc: + desc = self._first_paragraph(lines) + + content_outline = self._extract_outline(sections) + + return { + 'name': title, + 'description': self._clean_text(desc)[:240], + 'use_cases': self._clean_text(use_cases)[:240], + 'usage': self._clean_text(usage)[:600], + 'examples': self._clean_text(examples)[:600], + 'content_outline': content_outline, + 'source_url': source_url + } + except Exception: + return None + + def _extract_markdown_sections(self, lines): + sections = {} + current = '__intro__' + buffer = [] + + for raw_line in lines: + line = raw_line.strip() + if line.startswith('##'): + sections[current] = '\n'.join(buffer).strip() + current = re.sub(r'^#+\s*', '', line).strip().lower() + buffer = [] + else: + buffer.append(raw_line) + + sections[current] = '\n'.join(buffer).strip() + return sections + + def _pick_section(self, sections, keywords): + for key, value in sections.items(): + low = key.lower() + for keyword in keywords: + if keyword.lower() in low and value.strip(): + return value.strip() + return '' + + def _first_paragraph(self, lines): + paragraph = [] + for line in lines: + stripped = line.strip() + if not stripped or stripped.startswith('#'): + if paragraph: + break + continue + if stripped.startswith('- ') or stripped.startswith('* ') or stripped.startswith('```'): + continue + paragraph.append(stripped) + if len(' '.join(paragraph)) > 220: + break + return ' '.join(paragraph) + + def _extract_outline(self, sections): + outlines = [] + for key in sections.keys(): + if key != '__intro__': + clean_key = key.strip() + if clean_key: + outlines.append(clean_key) + return outlines[:8] + + def _clean_text(self, text): + if not text: + return '' + text = re.sub(r'```.*?```', '', text, flags=re.S) + text = re.sub(r'`([^`]+)`', r'\1', text) + text = re.sub(r'\[(.*?)\]\((.*?)\)', r'\1', text) + text = re.sub(r'^[-*]\s+', '', text, flags=re.M) + text = re.sub(r'\n{2,}', '\n', text) + text = re.sub(r'\s+', ' ', text) + return text.strip() + + def _summarize_skills(self, repo_info, skills): + if not skills: + return { + 'name': repo_info.get('name', ''), + 'short_desc': repo_info.get('description', ''), + 'description': repo_info.get('description', ''), + 'usage': '', + 'examples': '' + } + + skill_names = [s['name'] for s in skills if s.get('name')] + use_cases = [s['use_cases'] for s in skills if s.get('use_cases')] + descs = [s['description'] for s in skills if s.get('description')] + outlines = [] + for s in skills: + outlines.extend(s.get('content_outline', [])) + + repo_desc = (repo_info.get('description') or '').strip() + short_desc = repo_desc or (descs[0] if descs else '') + + parts = [] + if repo_desc: + parts.append(f"仓库简介:{repo_desc}") + if skill_names: + shown = '、'.join(skill_names[:5]) + more = f" 等 {len(skill_names)} 个 skill" if len(skill_names) > 5 else '' + parts.append(f"包含内容:{shown}{more}") + if use_cases: + parts.append(f"应用场景:{use_cases[0]}") + elif descs: + parts.append(f"应用场景:{descs[0]}") + if outlines: + uniq = [] + for item in outlines: + if item not in uniq: + uniq.append(item) + parts.append(f"文档涵盖:{'、'.join(uniq[:6])}") + + usage_lines = [] + for s in skills[:5]: + line = f"- {s['name']}" + if s.get('use_cases'): + line += f":{s['use_cases']}" + elif s.get('description'): + line += f":{s['description']}" + usage_lines.append(line) + + examples_lines = [] + for s in skills[:3]: + if s.get('source_url'): + examples_lines.append(f"{s['name']}:{s['source_url']}") + + return { + 'name': repo_info.get('name', ''), + 'short_desc': short_desc[:180], + 'description': '\n'.join(parts)[:1200], + 'usage': '\n'.join(usage_lines)[:1200], + 'examples': '\n'.join(examples_lines)[:1200] + } + def _extract_title(self, soup): """提取网站标题""" # 优先使用 og:title @@ -80,89 +318,83 @@ class WebsiteFetcher: if meta_desc and meta_desc.get('content'): return meta_desc['content'].strip() - # 使用 meta keywords 作为fallback - meta_keywords = soup.find('meta', attrs={'name': 'keywords'}) - if meta_keywords and meta_keywords.get('content'): - return meta_keywords['content'].strip() - return '' def _extract_logo(self, soup, base_url): - """提取网站Logo""" - logo_url = None + """提取网站 Logo URL""" + # 优先查找 favicon + favicon = soup.find('link', rel='icon') or soup.find('link', rel='shortcut icon') + if favicon and favicon.get('href'): + return urljoin(base_url, favicon['href']) - # 1. 尝试 og:image - og_image = soup.find('meta', property='og:image') - if og_image and og_image.get('content'): - logo_url = og_image['content'] + # 查找 apple-touch-icon + apple_icon = soup.find('link', rel='apple-touch-icon') + if apple_icon and apple_icon.get('href'): + return urljoin(base_url, apple_icon['href']) - # 2. 尝试 link rel="icon" 或 "shortcut icon" - if not logo_url: - icon_link = soup.find('link', rel=lambda x: x and ('icon' in x.lower() if isinstance(x, str) else 'icon' in ' '.join(x).lower())) - if icon_link and icon_link.get('href'): - logo_url = icon_link['href'] + # 使用 /favicon.ico + parsed = urlparse(base_url) + return f"{parsed.scheme}://{parsed.netloc}/favicon.ico" - # 3. 尝试 apple-touch-icon - if not logo_url: - apple_icon = soup.find('link', rel='apple-touch-icon') - if apple_icon and apple_icon.get('href'): - logo_url = apple_icon['href'] - - # 4. 默认使用 /favicon.ico - if not logo_url: - logo_url = '/favicon.ico' - - # 转换为绝对URL - if logo_url: - logo_url = urljoin(base_url, logo_url) - - return logo_url - - def download_logo(self, logo_url, save_dir='static/uploads'): + def download_logo(self, logo_url, save_dir='static/logos'): """ - 下载并保存Logo + 下载 Logo 到本地 Args: - logo_url: Logo的URL + logo_url: Logo URL save_dir: 保存目录 Returns: - str: 保存后的相对路径,失败返回None + str: 保存后的文件路径,失败返回 None """ - if not logo_url: - return None - try: - # 创建保存目录 + # 确保目录存在 os.makedirs(save_dir, exist_ok=True) # 下载图片 - response = requests.get(logo_url, headers=self.headers, timeout=self.timeout) - response.raise_for_status() + response = requests.get(logo_url, headers=self.headers, timeout=10, stream=True) + if response.status_code != 200: + return None # 检查是否是图片 - content_type = response.headers.get('content-type', '') - if not content_type.startswith('image/'): + content_type = response.headers.get('Content-Type', '').lower() + if 'image' not in content_type: return None # 生成文件名 - parsed_url = urlparse(logo_url) - ext = os.path.splitext(parsed_url.path)[1] - if not ext or len(ext) > 5: - ext = '.png' # 默认扩展名 + import time + import hashlib + ext = '.png' + if 'jpeg' in content_type or 'jpg' in content_type: + ext = '.jpg' + elif 'gif' in content_type: + ext = '.gif' + elif 'svg' in content_type: + ext = '.svg' + elif 'ico' in content_type: + ext = '.ico' - # 使用域名作为文件名 - domain = parsed_url.netloc.replace(':', '_').replace('.', '_') - filename = f"logo_{domain}{ext}" + timestamp = str(int(time.time() * 1000)) + hash_name = hashlib.md5(f"{logo_url}{timestamp}".encode()).hexdigest()[:16] + filename = f"logo_{hash_name}{ext}" filepath = os.path.join(save_dir, filename) # 保存图片 with open(filepath, 'wb') as f: - f.write(response.content) + for chunk in response.iter_content(8192): + f.write(chunk) - # 返回相对路径(用于数据库存储) - return f'/{filepath.replace(os.sep, "/")}' + # 验证图片是否有效 + try: + with Image.open(filepath) as img: + img.verify() + return filepath + except: + # 图片无效,删除文件 + if os.path.exists(filepath): + os.remove(filepath) + return None except Exception as e: - print(f"下载Logo失败: {str(e)}") - return None + print(f"下载 Logo 失败: {str(e)}") + return None \ No newline at end of file