feat: 优化 Skills URL 自动识别

- 新增 /api/fetch-skill-info 接口识别 GitHub skills 仓库
- 解析 skill markdown 的简介、应用场景和包含内容
- Skills 创建页支持一键自动填充标题、简介、来源链接

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
Jowe
2026-03-25 23:11:34 +08:00
parent f767a8a1b4
commit 3a63bc907e
3 changed files with 563 additions and 56 deletions

51
app.py
View File

@@ -1909,6 +1909,56 @@ def create_app(config_name='default'):
'message': f'抓取失败: {str(e)}' 'message': f'抓取失败: {str(e)}'
}), 500 }), 500
@app.route('/api/fetch-skill-info', methods=['POST'])
@login_required
def fetch_skill_info():
"""抓取 Skill 信息GitHub 仓库)"""
# 只允许管理员访问
if not isinstance(current_user, AdminModel):
return jsonify({'success': False, 'message': '无权访问'}), 403
try:
data = request.get_json()
url = data.get('url', '').strip()
if not url:
return jsonify({
'success': False,
'message': '请提供 GitHub 仓库URL'
}), 400
# 创建抓取器
fetcher = WebsiteFetcher(timeout=15)
# 抓取 Skill 信息
info = fetcher.fetch_skill_info(url)
if not info:
return jsonify({
'success': False,
'message': '无法获取 Skill 信息,请检查 URL 是否为有效的 GitHub 仓库'
})
return jsonify({
'success': True,
'data': {
'name': info.get('name', ''),
'short_desc': info.get('short_desc', ''),
'description': info.get('description', ''),
'github_url': info.get('github_url', ''),
'source_repo': info.get('source_repo', ''),
'source_type': 'github',
'usage': info.get('usage', ''),
'examples': info.get('examples', '')
}
})
except Exception as e:
return jsonify({
'success': False,
'message': f'抓取失败: {str(e)}'
}), 500
@app.route('/api/upload-logo', methods=['POST']) @app.route('/api/upload-logo', methods=['POST'])
@login_required @login_required
def upload_logo(): def upload_logo():
@@ -3417,6 +3467,7 @@ Sitemap: {}sitemap.xml
# Skills管理视图 # Skills管理视图
class SkillAdmin(SecureModelView): class SkillAdmin(SecureModelView):
create_template = 'admin/skill/create.html'
can_edit = True can_edit = True
can_delete = True can_delete = True
can_create = True can_create = True

View File

@@ -0,0 +1,224 @@
{% extends 'admin/model/create.html' %}
{% block tail %}
{{ super() }}
<style>
.auto-fetch-btn {
margin-top: 10px;
margin-bottom: 15px;
}
.fetch-status {
margin-top: 10px;
padding: 10px;
border-radius: 8px;
display: none;
}
.fetch-status.success {
background-color: rgba(34, 197, 94, 0.1);
border: 1px solid rgba(34, 197, 94, 0.3);
color: #4ade80;
}
.fetch-status.error {
background-color: rgba(239, 68, 68, 0.1);
border: 1px solid rgba(239, 68, 68, 0.3);
color: #f87171;
}
.logo-preview {
margin-top: 10px;
max-width: 100px;
max-height: 100px;
display: none;
}
.logo-preview img {
max-width: 100%;
max-height: 80px;
object-fit: contain;
border-radius: 8px;
border: 1px solid #ddd;
}
</style>
{% endblock %}
{% block header %}
<h2>添加 Skill</h2>
{% endblock %}
{% block form %}
<div class="form-group">
<label for="github_url">GitHub 仓库 URL <span style="color: #999; font-weight: normal;">(可选,填写后自动获取信息)</span></label>
<div style="display: flex; gap: 10px; align-items: flex-start;">
<input type="text" class="form-control" id="github_url" placeholder="例如: https://github.com/anthropics/claude-code" style="flex: 1;">
<button type="button" class="auto-fetch-btn btn btn-info" id="fetchSkillBtn">
<span class="material-symbols-outlined" style="vertical-align: middle;">auto_awesome</span>
自动识别
</button>
</div>
<div class="fetch-status" id="fetchStatus"></div>
</div>
<div class="form-group">
<label for="name">Skill 名称 <span style="color: #e74c3c;">*</span></label>
<input type="text" class="form-control" name="name" id="name" required placeholder="例如: PDF Helper">
</div>
<div class="form-group">
<label for="short_desc">简短描述</label>
<input type="text" class="form-control" name="short_desc" id="short_desc" placeholder="一句话描述这个 Skill 的用途">
</div>
<div class="form-group">
<label for="description">详细介绍</label>
<textarea class="form-control" name="description" id="description" rows="4" placeholder="详细介绍这个 Skill 的功能和使用场景"></textarea>
</div>
<div class="form-group">
<label for="logo">Logo 图片路径</label>
<input type="text" class="form-control" name="logo" id="logo" placeholder="/static/logos/xxx.png">
<div class="logo-preview" id="logoPreview">
<img id="logoImg" src="" alt="Logo预览">
</div>
</div>
<div class="form-group">
<label for="github_url_field">GitHub 链接</label>
<input type="text" class="form-control" name="github_url" id="github_url_field" placeholder="https://github.com/...">
</div>
<div class="form-group">
<label for="source_type">来源类型</label>
<select class="form-control" name="source_type" id="source_type">
<option value="manual" selected>手动添加</option>
<option value="github">GitHub</option>
</select>
</div>
<div class="form-group">
<label for="source_repo">来源仓库</label>
<input type="text" class="form-control" name="source_repo" id="source_repo" placeholder="owner/repo">
</div>
<div class="form-group">
<label for="usage">包含内容 / 应用场景</label>
<textarea class="form-control" name="usage" id="usage" rows="4" placeholder="这个 skills 仓库主要包含哪些内容、适合什么场景"></textarea>
</div>
<div class="form-group">
<label for="examples">来源链接 / 示例</label>
<textarea class="form-control" name="examples" id="examples" rows="4" placeholder="主要来源链接或示例链接"></textarea>
</div>
<div class="form-group">
<div class="form-check">
<input type="checkbox" class="form-check-input" name="is_active" id="is_active" checked>
<label class="form-check-label" for="is_active">启用</label>
</div>
</div>
<div class="form-group">
<div class="form-check">
<input type="checkbox" class="form-check-input" name="is_featured" id="is_featured">
<label class="form-check-label" for="is_featured">推荐</label>
</div>
</div>
<div class="form-group">
<label for="sort_order">排序权重</label>
<input type="number" class="form-control" name="sort_order" id="sort_order" value="0" placeholder="数字越大越靠前">
</div>
<div class="form-group">
<button type="submit" class="btn btn-primary">保存</button>
<a href="{{ url_for('skills.index_view') }}" class="btn btn-secondary">取消</a>
</div>
{% endblock %}
{% block tail2 %}
<script>
document.addEventListener('DOMContentLoaded', function() {
const fetchBtn = document.getElementById('fetchSkillBtn');
const urlInput = document.getElementById('github_url');
const statusDiv = document.getElementById('fetchStatus');
const logoInput = document.getElementById('logo');
const logoPreview = document.getElementById('logoPreview');
const logoImg = document.getElementById('logoImg');
function showStatus(message, type) {
statusDiv.textContent = message;
statusDiv.className = 'fetch-status ' + type;
statusDiv.style.display = 'block';
}
if (logoInput) {
logoInput.addEventListener('input', function() {
if (this.value) {
logoImg.src = this.value;
logoPreview.style.display = 'block';
} else {
logoPreview.style.display = 'none';
}
});
}
if (fetchBtn && urlInput) {
fetchBtn.addEventListener('click', function() {
const url = urlInput.value.trim();
if (!url) {
showStatus('请先输入 GitHub 仓库 URL', 'error');
return;
}
if (!url.includes('github.com')) {
showStatus('请输入有效的 GitHub 仓库 URL', 'error');
return;
}
fetchBtn.disabled = true;
statusDiv.style.display = 'none';
fetch('/api/fetch-skill-info', {
method: 'POST',
headers: {
'Content-Type': 'application/json',
},
body: JSON.stringify({ url: url })
})
.then(response => response.json())
.then(data => {
if (!data.success) {
showStatus('✗ ' + (data.message || '获取失败,请手动填写'), 'error');
return;
}
const nameField = document.querySelector('input[name="name"]');
const shortDescField = document.querySelector('input[name="short_desc"]');
const descriptionField = document.querySelector('textarea[name="description"]');
const githubUrlField = document.querySelector('input[name="github_url"]');
const sourceRepoField = document.querySelector('input[name="source_repo"]');
const sourceTypeField = document.querySelector('select[name="source_type"]');
const usageField = document.querySelector('textarea[name="usage"]');
const examplesField = document.querySelector('textarea[name="examples"]');
if (nameField && data.data.name) nameField.value = data.data.name;
if (shortDescField && data.data.short_desc) shortDescField.value = data.data.short_desc.substring(0, 100);
if (descriptionField && data.data.description) descriptionField.value = data.data.description;
if (githubUrlField && data.data.github_url) githubUrlField.value = data.data.github_url;
if (sourceRepoField && data.data.source_repo) sourceRepoField.value = data.data.source_repo;
if (sourceTypeField && data.data.source_type) sourceTypeField.value = data.data.source_type;
if (usageField && data.data.usage) usageField.value = data.data.usage;
if (examplesField && data.data.examples) examplesField.value = data.data.examples;
showStatus('✓ Skill 信息获取成功!已自动填充标题、简介、应用场景和来源链接', 'success');
})
.catch(error => {
console.error('Error:', error);
showStatus('✗ 网络请求失败,请手动填写', 'error');
})
.finally(() => {
fetchBtn.disabled = false;
});
});
}
});
</script>
{% endblock %}

View File

@@ -6,6 +6,7 @@ import requests
from bs4 import BeautifulSoup from bs4 import BeautifulSoup
from urllib.parse import urljoin, urlparse from urllib.parse import urljoin, urlparse
import os import os
import re
from PIL import Image from PIL import Image
from io import BytesIO from io import BytesIO
@@ -54,6 +55,243 @@ class WebsiteFetcher:
print(f"抓取网站信息失败: {str(e)}") print(f"抓取网站信息失败: {str(e)}")
return None return None
def fetch_skill_info(self, url):
"""
抓取 Skill 信息GitHub 仓库)
Args:
url: GitHub 仓库URL
Returns:
dict: 包含name, short_desc, github_url, source_repo的字典
"""
try:
# 解析 GitHub URL
# 支持格式: https://github.com/user/repo 或 https://github.com/user/repo/tree/main
parsed = urlparse(url)
if 'github.com' not in parsed.netloc:
return None
path_parts = [p for p in parsed.path.split('/') if p]
if len(path_parts) < 2:
return None
owner = path_parts[0]
repo = path_parts[1].replace('.git', '')
# 获取仓库信息
api_url = f"https://api.github.com/repos/{owner}/{repo}"
response = requests.get(api_url, headers=self.headers, timeout=self.timeout)
if response.status_code != 200:
return None
repo_info = response.json()
# 获取 .claude 目录下的 skills
skills_url = f"https://api.github.com/repos/{owner}/{repo}/contents/.claude/skills"
skills_response = requests.get(skills_url, headers=self.headers, timeout=self.timeout)
skills = []
if skills_response.status_code == 200:
skills_data = skills_response.json()
for item in skills_data:
if item.get('type') == 'file' and item['name'].endswith('.md'):
# 获取 skill 文件内容
skill_content = requests.get(item['download_url'], headers=self.headers, timeout=self.timeout)
if skill_content.status_code == 200:
skill_info = self._parse_skill_file(
item['name'],
skill_content.text,
item.get('html_url') or item.get('download_url') or ''
)
if skill_info:
skills.append(skill_info)
summary = self._summarize_skills(repo_info, skills)
return {
'name': summary.get('name') or repo_info.get('name', ''),
'short_desc': summary.get('short_desc') or repo_info.get('description', ''),
'description': summary.get('description', ''),
'github_url': url,
'source_repo': f"{owner}/{repo}",
'usage': summary.get('usage', ''),
'examples': summary.get('examples', ''),
'skills': skills
}
except Exception as e:
print(f"抓取 Skill 信息失败: {str(e)}")
return None
def _parse_skill_file(self, filename, content, source_url=''):
"""解析 skill 文件内容"""
try:
lines = [line.rstrip() for line in content.strip().split('\n')]
fallback_name = filename.replace('.md', '')
title = fallback_name
for line in lines:
if line.startswith('# '):
title = line[2:].strip()
break
sections = self._extract_markdown_sections(lines)
desc = self._pick_section(
sections,
['描述', '简介', 'description', 'overview', 'summary', 'what it does']
)
use_cases = self._pick_section(
sections,
['应用场景', '适用场景', '使用场景', 'use cases', 'when to use', 'use case']
)
usage = self._pick_section(
sections,
['使用方法', '用法', 'usage', 'how to use', 'quick start']
)
examples = self._pick_section(
sections,
['示例', 'examples', 'example']
)
if not desc:
desc = self._first_paragraph(lines)
content_outline = self._extract_outline(sections)
return {
'name': title,
'description': self._clean_text(desc)[:240],
'use_cases': self._clean_text(use_cases)[:240],
'usage': self._clean_text(usage)[:600],
'examples': self._clean_text(examples)[:600],
'content_outline': content_outline,
'source_url': source_url
}
except Exception:
return None
def _extract_markdown_sections(self, lines):
sections = {}
current = '__intro__'
buffer = []
for raw_line in lines:
line = raw_line.strip()
if line.startswith('##'):
sections[current] = '\n'.join(buffer).strip()
current = re.sub(r'^#+\s*', '', line).strip().lower()
buffer = []
else:
buffer.append(raw_line)
sections[current] = '\n'.join(buffer).strip()
return sections
def _pick_section(self, sections, keywords):
for key, value in sections.items():
low = key.lower()
for keyword in keywords:
if keyword.lower() in low and value.strip():
return value.strip()
return ''
def _first_paragraph(self, lines):
paragraph = []
for line in lines:
stripped = line.strip()
if not stripped or stripped.startswith('#'):
if paragraph:
break
continue
if stripped.startswith('- ') or stripped.startswith('* ') or stripped.startswith('```'):
continue
paragraph.append(stripped)
if len(' '.join(paragraph)) > 220:
break
return ' '.join(paragraph)
def _extract_outline(self, sections):
outlines = []
for key in sections.keys():
if key != '__intro__':
clean_key = key.strip()
if clean_key:
outlines.append(clean_key)
return outlines[:8]
def _clean_text(self, text):
if not text:
return ''
text = re.sub(r'```.*?```', '', text, flags=re.S)
text = re.sub(r'`([^`]+)`', r'\1', text)
text = re.sub(r'\[(.*?)\]\((.*?)\)', r'\1', text)
text = re.sub(r'^[-*]\s+', '', text, flags=re.M)
text = re.sub(r'\n{2,}', '\n', text)
text = re.sub(r'\s+', ' ', text)
return text.strip()
def _summarize_skills(self, repo_info, skills):
if not skills:
return {
'name': repo_info.get('name', ''),
'short_desc': repo_info.get('description', ''),
'description': repo_info.get('description', ''),
'usage': '',
'examples': ''
}
skill_names = [s['name'] for s in skills if s.get('name')]
use_cases = [s['use_cases'] for s in skills if s.get('use_cases')]
descs = [s['description'] for s in skills if s.get('description')]
outlines = []
for s in skills:
outlines.extend(s.get('content_outline', []))
repo_desc = (repo_info.get('description') or '').strip()
short_desc = repo_desc or (descs[0] if descs else '')
parts = []
if repo_desc:
parts.append(f"仓库简介:{repo_desc}")
if skill_names:
shown = ''.join(skill_names[:5])
more = f"{len(skill_names)} 个 skill" if len(skill_names) > 5 else ''
parts.append(f"包含内容:{shown}{more}")
if use_cases:
parts.append(f"应用场景:{use_cases[0]}")
elif descs:
parts.append(f"应用场景:{descs[0]}")
if outlines:
uniq = []
for item in outlines:
if item not in uniq:
uniq.append(item)
parts.append(f"文档涵盖:{''.join(uniq[:6])}")
usage_lines = []
for s in skills[:5]:
line = f"- {s['name']}"
if s.get('use_cases'):
line += f"{s['use_cases']}"
elif s.get('description'):
line += f"{s['description']}"
usage_lines.append(line)
examples_lines = []
for s in skills[:3]:
if s.get('source_url'):
examples_lines.append(f"{s['name']}{s['source_url']}")
return {
'name': repo_info.get('name', ''),
'short_desc': short_desc[:180],
'description': '\n'.join(parts)[:1200],
'usage': '\n'.join(usage_lines)[:1200],
'examples': '\n'.join(examples_lines)[:1200]
}
def _extract_title(self, soup): def _extract_title(self, soup):
"""提取网站标题""" """提取网站标题"""
# 优先使用 og:title # 优先使用 og:title
@@ -80,88 +318,82 @@ class WebsiteFetcher:
if meta_desc and meta_desc.get('content'): if meta_desc and meta_desc.get('content'):
return meta_desc['content'].strip() return meta_desc['content'].strip()
# 使用 meta keywords 作为fallback
meta_keywords = soup.find('meta', attrs={'name': 'keywords'})
if meta_keywords and meta_keywords.get('content'):
return meta_keywords['content'].strip()
return '' return ''
def _extract_logo(self, soup, base_url): def _extract_logo(self, soup, base_url):
"""提取网站Logo""" """提取网站 Logo URL"""
logo_url = None # 优先查找 favicon
favicon = soup.find('link', rel='icon') or soup.find('link', rel='shortcut icon')
if favicon and favicon.get('href'):
return urljoin(base_url, favicon['href'])
# 1. 尝试 og:image # 查找 apple-touch-icon
og_image = soup.find('meta', property='og:image')
if og_image and og_image.get('content'):
logo_url = og_image['content']
# 2. 尝试 link rel="icon" 或 "shortcut icon"
if not logo_url:
icon_link = soup.find('link', rel=lambda x: x and ('icon' in x.lower() if isinstance(x, str) else 'icon' in ' '.join(x).lower()))
if icon_link and icon_link.get('href'):
logo_url = icon_link['href']
# 3. 尝试 apple-touch-icon
if not logo_url:
apple_icon = soup.find('link', rel='apple-touch-icon') apple_icon = soup.find('link', rel='apple-touch-icon')
if apple_icon and apple_icon.get('href'): if apple_icon and apple_icon.get('href'):
logo_url = apple_icon['href'] return urljoin(base_url, apple_icon['href'])
# 4. 默认使用 /favicon.ico # 使用 /favicon.ico
if not logo_url: parsed = urlparse(base_url)
logo_url = '/favicon.ico' return f"{parsed.scheme}://{parsed.netloc}/favicon.ico"
# 转换为绝对URL def download_logo(self, logo_url, save_dir='static/logos'):
if logo_url:
logo_url = urljoin(base_url, logo_url)
return logo_url
def download_logo(self, logo_url, save_dir='static/uploads'):
""" """
下载并保存Logo 下载 Logo 到本地
Args: Args:
logo_url: LogoURL logo_url: Logo URL
save_dir: 保存目录 save_dir: 保存目录
Returns: Returns:
str: 保存后的相对路径失败返回None str: 保存后的文件路径,失败返回 None
""" """
if not logo_url:
return None
try: try:
# 创建保存目录 # 确保目录存在
os.makedirs(save_dir, exist_ok=True) os.makedirs(save_dir, exist_ok=True)
# 下载图片 # 下载图片
response = requests.get(logo_url, headers=self.headers, timeout=self.timeout) response = requests.get(logo_url, headers=self.headers, timeout=10, stream=True)
response.raise_for_status() if response.status_code != 200:
return None
# 检查是否是图片 # 检查是否是图片
content_type = response.headers.get('content-type', '') content_type = response.headers.get('Content-Type', '').lower()
if not content_type.startswith('image/'): if 'image' not in content_type:
return None return None
# 生成文件名 # 生成文件名
parsed_url = urlparse(logo_url) import time
ext = os.path.splitext(parsed_url.path)[1] import hashlib
if not ext or len(ext) > 5: ext = '.png'
ext = '.png' # 默认扩展名 if 'jpeg' in content_type or 'jpg' in content_type:
ext = '.jpg'
elif 'gif' in content_type:
ext = '.gif'
elif 'svg' in content_type:
ext = '.svg'
elif 'ico' in content_type:
ext = '.ico'
# 使用域名作为文件名 timestamp = str(int(time.time() * 1000))
domain = parsed_url.netloc.replace(':', '_').replace('.', '_') hash_name = hashlib.md5(f"{logo_url}{timestamp}".encode()).hexdigest()[:16]
filename = f"logo_{domain}{ext}" filename = f"logo_{hash_name}{ext}"
filepath = os.path.join(save_dir, filename) filepath = os.path.join(save_dir, filename)
# 保存图片 # 保存图片
with open(filepath, 'wb') as f: with open(filepath, 'wb') as f:
f.write(response.content) for chunk in response.iter_content(8192):
f.write(chunk)
# 返回相对路径(用于数据库存储) # 验证图片是否有效
return f'/{filepath.replace(os.sep, "/")}' try:
with Image.open(filepath) as img:
img.verify()
return filepath
except:
# 图片无效,删除文件
if os.path.exists(filepath):
os.remove(filepath)
return None
except Exception as e: except Exception as e:
print(f"下载 Logo 失败: {str(e)}") print(f"下载 Logo 失败: {str(e)}")