Merge branch 'master' of https://gitlink.org.cn/whale_hihihi/gitlink-cli into surponess_br

# Conflicts:
#	.gitignore
#	gitlink-cli.exe
This commit is contained in:
Surponess 2026-07-06 11:19:32 +08:00
commit 4e8b49444e
51 changed files with 8853 additions and 179 deletions

View File

@ -20,7 +20,7 @@ workflow:
ssh_ip: '"121.41.222.73"'
ssh_port: '"22"'
ssh_user: '"root"'
ssh_cmd: '"if [ -d /root/gitlink-cli/.git ]; then cd /root/gitlink-cli && git fetch origin && git checkout master && git reset --hard origin/master; else rm -rf /root/gitlink-cli && git clone https://gitlink.org.cn/whale_hihihi/gitlink-cli.git /root/gitlink-cli && cd /root/gitlink-cli && git checkout master; fi && docker stop gitlink-cli 2>/dev/null; docker rm gitlink-cli 2>/dev/null; docker rmi gitlink-cli:latest 2>/dev/null; docker build --no-cache -t gitlink-cli:latest . && docker run -d --name gitlink-cli gitlink-cli:latest && echo Deploy success"'
ssh_cmd: '"if [ -d /root/gitlink-cli/.git ]; then cd /root/gitlink-cli && git fetch origin && git checkout master && git reset --hard origin/master; else rm -rf /root/gitlink-cli && git clone https://gitlink.org.cn/whale_hihihi/gitlink-cli.git /root/gitlink-cli && cd /root/gitlink-cli && git checkout master; fi && docker stop gitlink-cli 2>/dev/null; docker rm gitlink-cli 2>/dev/null; docker rmi gitlink-cli:latest 2>/dev/null; docker build --no-cache -t gitlink-cli:latest . && docker run -d --name gitlink-cli -p 8080:8080 -e GITLINK_TOKEN=((gitlink_token.token)) gitlink-cli:latest && echo Deploy success"'
needs:
- start
- ref: ssh_cmd_1

2
.gitignore vendored
View File

@ -1,6 +1,6 @@
docs/
doc/
# Pythondemo 网页后端)
# Pythondemo 网页后端)
__pycache__/
*.pyc
demo/bin/

View File

@ -1,13 +1,36 @@
# ============================================================
# 多阶段构建gitlink-cli 子赛题四网页终端
# ============================================================
# 阶段1 builder —— Go 静态编译
# ============================================================
FROM golang:1.26-alpine AS builder
ENV GOPROXY=https://goproxy.cn,direct
WORKDIR /app
WORKDIR /src
# 先拷依赖清单,利用 Docker 层缓存
COPY go.mod go.sum ./
RUN go mod download
COPY . .
RUN CGO_ENABLED=0 go build -ldflags="-s -w" -o gitlink-cli .
FROM alpine:latest
RUN apk --no-cache add ca-certificates git
COPY . .
# modernc.org/sqlite 是 pure-GoCGO_ENABLED=0 即可编译纯静态二进制
RUN CGO_ENABLED=0 GOOS=linux go build -ldflags="-s -w" -o /out/gitlink-cli .
# ============================================================
# 阶段2 runtime —— Python + Go 二进制
# ============================================================
FROM python:3.12-slim
# Go CLI 放入 PATH
COPY --from=builder /out/gitlink-cli /usr/local/bin/gitlink-cli
# 科研算法层:先拷 requirements.txt 安装依赖(利用层缓存),再拷源码
COPY scripts/research/requirements.txt /app/scripts/research/requirements.txt
RUN pip install --no-cache-dir -r /app/scripts/research/requirements.txt
COPY scripts/research/ /app/scripts/research/
WORKDIR /app
COPY --from=builder /app/gitlink-cli /usr/local/bin/gitlink-cli
ENTRYPOINT ["gitlink-cli"]
# 子赛题四网页终端 HTTP 服务
EXPOSE 8080
ENTRYPOINT ["gitlink-cli", "server", "--port", "8080", "--research-dir", "/app/scripts/research", "--work-dir", "/app/research-output"]

View File

@ -104,7 +104,7 @@ The official [GitLink](https://www.gitlink.org.cn) CLI tool — built for humans
| Category | Capabilities |
|----------|-------------|
| 📦 Repo | List, create, fork, delete repositories, view repo info, insights, and interactions |
| 🐛 Issue | Create, update, close, batch close, comment on issues |
| 🐛 Issue | Create, update, close, batch close/update/delete, comment on issues |
| 🔖 Label | Create, list, update, delete issue labels |
| 🔀 PR | Create, merge, review pull requests, view changed files |
| 👥 Member | List, add, remove repository members, change roles, create and accept invite links |
@ -114,6 +114,7 @@ The official [GitLink](https://www.gitlink.org.cn) CLI tool — built for humans
| 🔧 CI | View builds, logs, CI/CD operations |
| ⚙️ Pipeline | Run, inspect, enable, disable, delete pipeline workflows and logs |
| 🔔 Webhook | Manage repo webhooks and test deliveries |
| 📖 Wiki | List, view, create, update, and delete wiki pages |
| 🔍 Search | Search repositories, users |
| 📊 Dataset | Query research datasets by project |
| 👤 User | View user profiles and info |
@ -273,6 +274,28 @@ gitlink-cli webhook +test --owner Gitlink --repo forgeplus --id 68
gitlink-cli webhook +tasks --owner Gitlink --repo forgeplus --id 68
```
### Wiki Management
```bash
# List wiki pages (table of contents)
gitlink-cli wiki +list --owner Gitlink --repo forgeplus --project-id 12345
# View a wiki page by page name
gitlink-cli wiki +view --owner Gitlink --repo forgeplus --project-id 12345 -n home
# Create a wiki page
gitlink-cli wiki +create --owner Gitlink --repo forgeplus --project-id 12345 \
-n getting-started -t "Getting Started" -c "# Getting Started Guide"
# Update a wiki page title and/or content
gitlink-cli wiki +update --owner Gitlink --repo forgeplus --project-id 12345 -n home -t "New Title"
gitlink-cli wiki +update --owner Gitlink --repo forgeplus --project-id 12345 -n home -c "# Updated content"
gitlink-cli wiki +update --owner Gitlink --repo forgeplus --project-id 12345 -n home -t "New Title" -c "New content"
# Delete a wiki page
gitlink-cli wiki +delete --owner Gitlink --repo forgeplus --project-id 12345 -n old-page
```
### Member Management
```bash
@ -322,6 +345,14 @@ gitlink-cli issue +batch-close --owner Gitlink --repo forgeplus --numbers 123,12
# Batch close issues from a CSV file
gitlink-cli issue +batch-close --owner Gitlink --repo forgeplus --from issues.csv
# Preview batch metadata update by API issue IDs
# Note: --ids uses API issue IDs, not web URL issue numbers.
gitlink-cli issue +batch-update --owner Gitlink --repo forgeplus --ids 101,102 --status-id 3 --priority-id 2 --dry-run
# Destructive batch delete requires both dry-run first and --yes for real execution
gitlink-cli issue +batch-delete --owner Gitlink --repo forgeplus --ids 101,102 --dry-run
gitlink-cli issue +batch-delete --owner Gitlink --repo forgeplus --ids 101,102 --yes
# Add a comment
gitlink-cli issue +comment --owner Gitlink --repo forgeplus -i 123 -b "Fixed"
@ -707,7 +738,7 @@ See [skills/README.md](./skills/README.md) for details.
|-------|-------------|
| `gitlink-shared` | Authentication, global parameters, safety rules, API notes |
| `gitlink-repo` | Repository operations (create, view, delete, fork, insights, etc.) |
| `gitlink-issue` | Issue operations (create, update, close, comment, etc.) |
| `gitlink-issue` | Issue operations (create, update, close, batch update/delete, comment, etc.) |
| `gitlink-pr` | Pull request operations (create, merge, review, etc.) |
| `gitlink-member` | Repository member and invite link management |
| `gitlink-branch` | Branch management (create, delete, list, protect, unprotect) |

View File

@ -104,7 +104,7 @@
| 分类 | 能力 |
|------|------|
| 📦 仓库 | 列出、创建、Fork、删除仓库查看仓库信息、洞察数据和互动状态 |
| 🐛 Issue | 创建、更新、关闭、批量关闭、评论 Issue |
| 🐛 Issue | 创建、更新、关闭、批量关闭/更新/删除、评论 Issue |
| 🔖 标签 | 创建、列出、更新、删除 Issue 标签 |
| 🔀 PR | 创建、合并、Review Pull Request查看变更文件 |
| 👥 成员 | 列出、添加、移除仓库成员,调整角色,生成和接受邀请链接 |
@ -113,6 +113,7 @@
| 🏢 组织 | 管理组织、成员、团队 |
| 🔧 CI | 查看构建、日志、CI/CD 操作 |
| ⚙️ Pipeline | 运行、查看、启停、删除流水线工作流并查询日志 |
| 📖 Wiki | 列出、查看、创建、更新、删除 Wiki 页面 |
| 🔍 搜索 | 搜索仓库、用户 |
| 📊 数据集 | 按项目查询科研数据集 |
| 👤 用户 | 查看用户资料和信息 |
@ -284,6 +285,28 @@ gitlink-cli webhook +test --owner Gitlink --repo forgeplus --id 68
gitlink-cli webhook +tasks --owner Gitlink --repo forgeplus --id 68
```
### Wiki 管理
```bash
# 列出 Wiki 页面(目录结构)
gitlink-cli wiki +list --owner Gitlink --repo forgeplus --project-id 12345
# 查看 Wiki 页面
gitlink-cli wiki +view --owner Gitlink --repo forgeplus --project-id 12345 -n home
# 创建 Wiki 页面
gitlink-cli wiki +create --owner Gitlink --repo forgeplus --project-id 12345 \
-n getting-started -t "快速开始" -c "# 快速开始指南"
# 更新 Wiki 页面标题和/或内容
gitlink-cli wiki +update --owner Gitlink --repo forgeplus --project-id 12345 -n home -t "新标题"
gitlink-cli wiki +update --owner Gitlink --repo forgeplus --project-id 12345 -n home -c "# 更新后的内容"
gitlink-cli wiki +update --owner Gitlink --repo forgeplus --project-id 12345 -n home -t "新标题" -c "新内容"
# 删除 Wiki 页面
gitlink-cli wiki +delete --owner Gitlink --repo forgeplus --project-id 12345 -n old-page
```
### 成员管理
```bash
@ -333,6 +356,14 @@ gitlink-cli issue +batch-close --owner Gitlink --repo forgeplus --numbers 123,12
# 从 CSV 文件批量关闭 Issue
gitlink-cli issue +batch-close --owner Gitlink --repo forgeplus --from issues.csv
# 按 API issue id 预览批量更新元数据
# 注意:--ids 是 API issue id不是网页 URL 中的 Issue 编号。
gitlink-cli issue +batch-update --owner Gitlink --repo forgeplus --ids 101,102 --status-id 3 --priority-id 2 --dry-run
# 危险批量删除必须先 dry-run真实执行还要显式 --yes
gitlink-cli issue +batch-delete --owner Gitlink --repo forgeplus --ids 101,102 --dry-run
gitlink-cli issue +batch-delete --owner Gitlink --repo forgeplus --ids 101,102 --yes
# 添加评论
gitlink-cli issue +comment --owner Gitlink --repo forgeplus -i 123 -b "已修复"
@ -581,7 +612,7 @@ git push gitlink
|-------|------|
| `gitlink-shared` | 认证、全局参数、安全规则、API 注意事项 |
| `gitlink-repo` | 仓库操作创建、查看、删除、Fork、洞察数据等 |
| `gitlink-issue` | Issue 操作(创建、更新、关闭、评论等) |
| `gitlink-issue` | Issue 操作(创建、更新、关闭、批量更新/删除、评论等) |
| `gitlink-pr` | Pull Request 操作创建、合并、Review 等) |
| `gitlink-member` | 仓库成员与邀请链接管理 |
| `gitlink-release` | 发布管理(创建、编辑、更新、查看、删除等) |

View File

@ -14,6 +14,7 @@ import (
doctorCmd "github.com/gitlink-org/gitlink-cli/cmd/doctor"
internalConfig "github.com/gitlink-org/gitlink-cli/internal/config"
"github.com/gitlink-org/gitlink-cli/internal/i18n"
serverCmd "github.com/gitlink-org/gitlink-cli/cmd/server"
"github.com/gitlink-org/gitlink-cli/shortcuts"
)
@ -58,6 +59,7 @@ func NewRootCmd(opts RootOptions, tr *i18n.Translator) (*cobra.Command, error) {
rootCmd.AddCommand(apiCmd.NewAPICmd(tr))
rootCmd.AddCommand(configCmd.NewConfigCmd(tr))
rootCmd.AddCommand(doctorCmd.NewDoctorCmd(tr))
rootCmd.AddCommand(serverCmd.NewServerCmd())
rootCmd.AddCommand(newVersionCmd(version, tr))
shortcuts.RegisterAll(rootCmd, tr)

293
cmd/server/server.go Normal file
View File

@ -0,0 +1,293 @@
// Package server 提供子赛题四的「网页终端」演示服务子赛题四·Phase 7
//
// 在云服务器上 `gitlink-cli server` 启动一个轻量 HTTP 服务:前端按 7 个科研场景按钮
// 触发后端运行对应的 Python 算法脚本scripts/research/*.py并把 JSON / Mermaid /
// 报告等产物回显/渲染。所有数据仍由 gitlink-cli 获取,本服务只做命令编排与产物转发。
//
// 安全:仅接受 7 个固定场景 + owner/repo/keyword 结构化参数(非任意 shell
// 若设置 --token / DEMO_TOKEN 环境变量,则请求需带 X-Demo-Token 头匹配。
package server
import (
"embed"
"encoding/json"
"fmt"
"io/fs"
"net/http"
"os"
"os/exec"
"path/filepath"
"strings"
"sync"
"time"
"github.com/spf13/cobra"
)
//go:embed static/*
var staticFS embed.FS
const defaultPort = 8080
// 场景定义:前端按钮 ↔ 后端 Python 脚本。
type scenarioDef struct {
Key string `json:"key"`
Label string `json:"label"`
Desc string `json:"desc"`
Script string `json:"script"`
Needs []string `json:"needs"` // 需要的输入: "owner","repo","keyword"
Timeout int `json:"timeout_s"`
}
var scenarios = []scenarioDef{
{Key: "s1", Label: "S1 仓库洞悉", Desc: "科研项目演进谱系 + 创新点", Script: "lineage.py", Needs: []string{"owner", "repo"}, Timeout: 180},
{Key: "s2", Label: "S2 知识图谱", Desc: "科研领域知识图谱(networkx)", Script: "graph_build.py", Needs: []string{"keyword"}, Timeout: 240},
{Key: "s3", Label: "S3 合规复现", Desc: "许可证/密钥/复现性检查", Script: "repro.py", Needs: []string{"owner", "repo"}, Timeout: 120},
{Key: "s4", Label: "S4 协作匹配", Desc: "学者×缺口 智能匹配", Script: "match.py", Needs: []string{"owner", "repo"}, Timeout: 180},
{Key: "s5", Label: "S5 进度预警", Desc: "周报 + 风险预警", Script: "report.py", Needs: []string{"owner", "repo"}, Timeout: 180},
{Key: "s6", Label: "S6 成果可视化", Desc: "交互图表(plotly)", Script: "visual.py", Needs: []string{"owner", "repo"}, Timeout: 240},
{Key: "hotspot", Label: "🔥 热点追踪", Desc: "科研热点全景观测:飙升项目+活跃讨论+主题热度+学者团队", Script: "hotspot.py", Needs: []string{"keyword"}, Timeout: 300},
}
type Options struct {
Port int
ResearchDir string // scripts/research 目录
WorkDir string // 产物输出根目录
Token string // 可选鉴权 token
}
func NewServerCmd() *cobra.Command {
opts := Options{Port: defaultPort, ResearchDir: "scripts/research", WorkDir: "research-output"}
cmd := &cobra.Command{
Use: "server",
Short: "启动子赛题四网页终端HTTP 演示服务)",
RunE: func(cmd *cobra.Command, args []string) error {
return Run(opts)
},
}
cmd.Flags().IntVarP(&opts.Port, "port", "p", defaultPort, "监听端口")
cmd.Flags().StringVar(&opts.ResearchDir, "research-dir", "scripts/research", "scripts/research 目录")
cmd.Flags().StringVar(&opts.WorkDir, "work-dir", "research-output", "产物输出根目录")
cmd.Flags().StringVar(&opts.Token, "token", "", "可选鉴权 token亦可用 DEMO_TOKEN 环境变量)")
return cmd
}
// Run 启动 HTTP 服务(阻塞)。
func Run(opts Options) error {
if t := os.Getenv("DEMO_TOKEN"); t != "" && opts.Token == "" {
opts.Token = t
}
_ = os.MkdirAll(opts.WorkDir, 0o755)
mux := http.NewServeMux()
h := &handler{opts: opts}
mux.HandleFunc("GET /api/scenarios", h.handleScenarios)
mux.HandleFunc("POST /api/run", h.handleRun)
mux.HandleFunc("GET /api/result/{key}", h.handleResult)
mux.HandleFunc("GET /api/health", h.handleHealth)
sub, err := fs.Sub(staticFS, "static")
if err != nil {
return fmt.Errorf("static fs: %w", err)
}
mux.Handle("GET /", http.FileServer(http.FS(sub)))
addr := fmt.Sprintf(":%d", opts.Port)
fmt.Fprintf(os.Stderr, "子赛题四 网页终端已启动: http://localhost%s\n", addr)
fmt.Fprintf(os.Stderr, " research-dir=%s work-dir=%s auth=%v\n", opts.ResearchDir, opts.WorkDir, opts.Token != "")
srv := &http.Server{Addr: addr, Handler: mux, ReadHeaderTimeout: 10 * time.Second}
return srv.ListenAndServe()
}
type handler struct {
opts Options
mu sync.Mutex // 串行化场景执行,避免并发打爆 GitLink API
}
func (h *handler) authed(r *http.Request) bool {
if h.opts.Token == "" {
return true
}
return r.Header.Get("X-Demo-Token") == h.opts.Token
}
func (h *handler) handleHealth(w http.ResponseWriter, r *http.Request) {
writeJSON(w, map[string]any{"ok": true, "scenarios": len(scenarios)})
}
func (h *handler) handleScenarios(w http.ResponseWriter, r *http.Request) {
if !h.authed(r) {
http.Error(w, "unauthorized", http.StatusUnauthorized)
return
}
writeJSON(w, map[string]any{"ok": true, "scenarios": scenarios})
}
// handleResult 返回某场景最近一次运行的产物(供 result.html 独立结果页按 key 读取,
// URL 可刷新/分享,便于演示讲解)。无需鉴权串行锁——只读已落盘产物。
func (h *handler) handleResult(w http.ResponseWriter, r *http.Request) {
if !h.authed(r) {
http.Error(w, "unauthorized", http.StatusUnauthorized)
return
}
key := r.PathValue("key")
sc, ok := findScenario(key)
if !ok {
writeJSON(w, map[string]any{"ok": false, "error": "unknown scenario: " + key})
return
}
outDir := filepath.Join(h.opts.WorkDir, sc.Key)
writeJSON(w, map[string]any{
"ok": true,
"scenario": sc.Key,
"label": sc.Label,
"desc": sc.Desc,
"artifacts": readArtifacts(outDir),
})
}
type runRequest struct {
Scenario string `json:"scenario"`
Owner string `json:"owner"`
Repo string `json:"repo"`
Keyword string `json:"keyword"`
}
func (h *handler) handleRun(w http.ResponseWriter, r *http.Request) {
if !h.authed(r) {
http.Error(w, "unauthorized", http.StatusUnauthorized)
return
}
var req runRequest
if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
writeJSON(w, map[string]any{"ok": false, "error": "bad request: " + err.Error()})
return
}
sc, ok := findScenario(req.Scenario)
if !ok {
writeJSON(w, map[string]any{"ok": false, "error": "unknown scenario: " + req.Scenario})
return
}
for _, need := range sc.Needs {
if (need == "keyword" && req.Keyword == "") ||
(need == "owner" && req.Owner == "") ||
(need == "repo" && req.Repo == "") {
writeJSON(w, map[string]any{"ok": false, "error": "missing parameter: " + need})
return
}
}
// 串行执行:一次只跑一个场景,保护 GitLink API。
h.mu.Lock()
defer h.mu.Unlock()
scriptPath := filepath.Join(h.opts.ResearchDir, sc.Script)
outDir := filepath.Join(h.opts.WorkDir, sc.Key)
_ = os.MkdirAll(outDir, 0o755)
argv := []string{scriptPath, "--out", outDir}
if contains(sc.Needs, "owner") {
argv = append(argv, "--owner", req.Owner, "--repo", req.Repo)
}
if contains(sc.Needs, "keyword") {
argv = append(argv, "--keywords", req.Keyword)
}
// python3 优先,回退 python
py, err := pythonBin()
if err != nil {
writeJSON(w, map[string]any{"ok": false, "error": err.Error()})
return
}
cmd := exec.Command(py, argv...)
cmd.Env = os.Environ()
start := time.Now()
out, err := cmd.CombinedOutput()
dur := time.Since(start)
resp := map[string]any{
"ok": err == nil,
"scenario": sc.Key,
"command": py + " " + strings.Join(argv, " "),
"duration": dur.Truncate(time.Millisecond).String(),
"stdout": string(out),
"out_dir": outDir,
}
if err != nil {
resp["error"] = err.Error()
}
// 附带读取关键产物json + 第一个 mmd + report.md便于前端直接渲染
resp["artifacts"] = readArtifacts(outDir)
writeJSON(w, resp)
}
func findScenario(key string) (scenarioDef, bool) {
for _, s := range scenarios {
if s.Key == key || strings.EqualFold(s.Key, key) {
return s, true
}
}
return scenarioDef{}, false
}
func readArtifacts(dir string) map[string]string {
out := map[string]string{}
// json 产物(取第一个 *.json
if entries, err := os.ReadDir(dir); err == nil {
for _, e := range entries {
if e.IsDir() {
continue
}
name := e.Name()
switch {
case strings.HasSuffix(name, ".json"):
b, _ := os.ReadFile(filepath.Join(dir, name))
out["json"] = string(b)
case strings.HasSuffix(name, ".mmd"):
b, _ := os.ReadFile(filepath.Join(dir, name))
out["mermaid"] = string(b)
case name == "visual.html":
b, _ := os.ReadFile(filepath.Join(dir, name))
out["html"] = string(b)
case strings.HasSuffix(name, ".md"):
// 第一份 .md 报告report.md / weekly_report.md / compliance_report.md
if _, ok := out["report"]; !ok {
b, _ := os.ReadFile(filepath.Join(dir, name))
out["report"] = string(b)
}
}
}
}
return out
}
func pythonBin() (string, error) {
// 候选按 Linux 习惯 python3 优先,再 python / py(Windows)。
// 必须实测能产出Windows 的 WindowsApps\python3.exe 是 Store 桩,对 -c 也可能 exit 0 但不真正执行,
// 故用「stdout 必须含 PYOK」来拦截桩。
for _, name := range []string{"python3", "python", "py"} {
path, err := exec.LookPath(name)
if err != nil {
continue
}
if out, err := exec.Command(path, "-c", "print('PYOK')").Output(); err == nil &&
strings.Contains(string(out), "PYOK") {
return path, nil
}
}
return "", fmt.Errorf("python 未安装;容器需内置 python3 并 pip install -r scripts/research/requirements.txt")
}
func contains(xs []string, s string) bool {
for _, x := range xs {
if x == s {
return true
}
}
return false
}
func writeJSON(w http.ResponseWriter, v any) {
w.Header().Set("Content-Type", "application/json; charset=utf-8")
_ = json.NewEncoder(w).Encode(v)
}

View File

@ -0,0 +1,485 @@
<!DOCTYPE html>
<html lang="zh-CN">
<head>
<meta charset="UTF-8">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>科研热点追踪 · GitLink Research Atlas</title>
<link rel="preconnect" href="https://fonts.googleapis.com">
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
<link href="https://fonts.googleapis.com/css2?family=Cormorant+Garamond:wght@500;600;700&family=Inter:wght@400;500;600;700&family=JetBrains+Mono:wght@500;700&family=Noto+Serif+SC:wght@600;700&family=Noto+Sans+SC:wght@400;500;700&display=swap" rel="stylesheet">
<style>
:root{
--bg:#F7F9FC; --card:#FFFFFF; --ink:#1E2A44; --ink2:#475569; --mute:#94A3B8;
--line:#E6EBF2; --indigo:#4F6BED; --coral:#F26B5E; --mint:#2EC4B6;
--t-blue:#E8EEFF; --t-coral:#FFEDEA; --t-mint:#E6F7F4; --t-amber:#FEF3C7;
--serif:"Cormorant Garamond","Noto Serif SC",Georgia,serif;
--sans:"Inter","Noto Sans SC",-apple-system,"PingFang SC","Microsoft YaHei",sans-serif;
--mono:"JetBrains Mono",ui-monospace,SFMono-Regular,Menlo,Consolas,monospace;
--shadow:0 1px 2px rgba(30,42,68,.04),0 6px 18px rgba(30,42,68,.06);
--shadow-sm:0 1px 2px rgba(30,42,68,.06);
}
*{box-sizing:border-box;margin:0;padding:0}
html,body{min-height:100%}
body{font-family:var(--sans);color:var(--ink);background:var(--bg);
background-image:radial-gradient(1200px 480px at 80% -8%,#EAF0FF 0%,rgba(234,240,255,0) 60%),
radial-gradient(900px 420px at 0% 0%,#FFF1EE 0%,rgba(255,241,238,0) 55%);
-webkit-font-smoothing:antialiased}
.wrap{max-width:1180px;margin:0 auto;padding:28px 36px 72px}
/* header */
.hdr{display:flex;align-items:center;gap:14px;margin-bottom:22px}
.back{display:inline-flex;align-items:center;gap:6px;font-size:13px;font-weight:600;color:var(--indigo);
text-decoration:none;padding:7px 14px;border-radius:10px;background:var(--t-blue);white-space:nowrap}
.back:hover{background:#D6DEFF}
.hdr .titles h1{font-family:var(--serif);font-weight:700;font-size:30px;line-height:1.15}
.hdr .titles .sub{font-size:13px;color:var(--mute);margin-top:3px}
/* control panel */
.ctl{background:var(--card);border:1px solid var(--line);border-radius:16px;box-shadow:var(--shadow);
padding:22px 24px;margin-bottom:22px}
.row{display:flex;align-items:center;gap:12px;flex-wrap:wrap}
.row + .row{margin-top:14px}
.row .lbl{font-size:12px;font-weight:600;color:var(--ink2);min-width:84px;letter-spacing:.02em}
.search-input{flex:1;min-width:260px;display:flex;align-items:center;gap:8px;background:#F8FAFD;
border:1px solid var(--line);border-radius:11px;padding:11px 14px}
.search-input:focus-within{border-color:var(--indigo);background:#fff;box-shadow:0 0 0 3px rgba(79,107,237,.12)}
.search-input input{border:0;outline:0;font-family:var(--mono);font-size:14px;width:100%;background:transparent;color:var(--ink)}
.ex-chips{display:flex;gap:7px;flex-wrap:wrap}
.ex-chips .ex{font-size:12px;color:var(--ink2);background:#F1F5F9;border:1px solid var(--line);
padding:5px 11px;border-radius:999px;cursor:pointer;transition:.15s}
.ex-chips .ex:hover{background:var(--t-blue);border-color:#C7D2FE;color:var(--indigo)}
.angles{display:flex;gap:9px;flex-wrap:wrap}
.ang{display:inline-flex;align-items:center;gap:7px;font-size:13px;font-weight:500;padding:8px 14px;
border-radius:11px;border:1px solid var(--line);background:#fff;cursor:pointer;transition:.15s;user-select:none}
.ang .dot{width:9px;height:9px;border-radius:50%;background:var(--mute);transition:.15s}
.ang.on{border-color:var(--indigo);background:#F5F8FF;color:var(--ink)}
.ang.on .dot{background:var(--indigo);box-shadow:0 0 0 3px rgba(79,107,237,.18)}
.ang[data-a="trending"] .dot{background:var(--coral)} .ang.on[data-a="trending"] .dot{background:var(--coral);box-shadow:0 0 0 3px rgba(242,107,94,.2)}
.ang[data-a="active"] .dot{background:#F59E0B} .ang.on[data-a="active"] .dot{background:#F59E0B;box-shadow:0 0 0 3px rgba(245,158,11,.2)}
.ang[data-a="topics"] .dot{background:#06B6D4} .ang.on[data-a="topics"] .dot{background:#06B6D4;box-shadow:0 0 0 3px rgba(6,182,212,.2)}
.ang[data-a="scholars"] .dot{background:var(--mint)} .ang.on[data-a="scholars"] .dot{background:var(--mint);box-shadow:0 0 0 3px rgba(46,196,182,.2)}
select.lim{font-family:var(--mono);font-size:13px;padding:8px 10px;border:1px solid var(--line);border-radius:10px;background:#fff;color:var(--ink)}
.btn-run{margin-left:auto;font-size:14px;font-weight:600;color:#fff;background:linear-gradient(135deg,var(--indigo),#7C8CF5);
border:0;border-radius:11px;padding:11px 22px;cursor:pointer;box-shadow:var(--shadow-sm);transition:.15s}
.btn-run:hover{filter:brightness(1.05);transform:translateY(-1px)}
.btn-run:disabled{opacity:.6;cursor:not-allowed;transform:none}
.status{font-family:var(--mono);font-size:12px;color:var(--mute);margin-top:14px;display:flex;align-items:center;gap:8px}
.status .ok{color:var(--mint)} .status .err{color:var(--coral)}
.spin{width:14px;height:14px;border-radius:50%;border:2px solid var(--t-blue);border-top-color:var(--indigo);animation:rot .8s linear infinite}
@keyframes rot{to{transform:rotate(360deg)}}
/* summary banner */
.summary{display:grid;grid-template-columns:repeat(4,1fr);gap:12px;margin-bottom:18px}
.sum-tile{border:1px solid var(--line);border-radius:13px;padding:14px 16px;background:var(--card);box-shadow:var(--shadow-sm)}
.sum-tile .sl{font-size:11px;color:var(--mute);font-weight:600;letter-spacing:.02em}
.sum-tile .sv{font-family:var(--mono);font-weight:700;font-size:26px;line-height:1.1;margin-top:4px;color:var(--ink)}
.sum-tile .ss{font-size:11px;color:var(--mute);margin-top:2px}
/* sections */
.sec{background:var(--card);border:1px solid var(--line);border-radius:16px;box-shadow:var(--shadow);
padding:20px 24px;margin-top:18px}
.sec-head{display:flex;align-items:center;gap:10px;margin-bottom:16px}
.sec-head .ico{width:30px;height:30px;border-radius:9px;display:grid;place-items:center;font-size:15px}
.sec-head h2{font-family:var(--serif);font-weight:700;font-size:21px}
.sec-head .eyebrow{font-family:var(--mono);font-size:10px;letter-spacing:.16em;text-transform:uppercase;color:var(--mute);display:block;margin-top:1px}
.sec-head .cnt{font-family:var(--mono);font-size:12px;color:var(--mute);margin-left:auto}
.ic-coral{background:var(--t-coral);color:var(--coral)} .ic-amber{background:var(--t-amber);color:#B45309}
.ic-cyan{background:#E0F7FA;color:#0891B2} .ic-mint{background:var(--t-mint);color:#0E7E72}
/* trending table */
.tbl{width:100%;border-collapse:collapse;font-size:13px}
.tbl th{text-align:left;font-size:11px;color:var(--mute);font-weight:600;padding:8px 10px;border-bottom:1px solid var(--line);text-transform:uppercase;letter-spacing:.04em}
.tbl td{padding:10px;border-bottom:1px solid var(--line);vertical-align:top}
.tbl tr:hover td{background:#F8FAFD}
.tbl .repo{font-family:var(--mono);font-weight:600;color:var(--indigo)}
.tbl .num{font-family:var(--mono);font-weight:700;text-align:right}
.lang-pill{font-size:11px;padding:2px 8px;border-radius:6px;background:var(--t-blue);color:var(--indigo)}
.scorebar{height:6px;border-radius:4px;background:var(--line);overflow:hidden;margin-top:4px;min-width:70px}
.scorebar i{display:block;height:100%;background:linear-gradient(90deg,var(--coral),#FF9B8F)}
.kw-chip{display:inline-block;font-size:10px;font-weight:600;padding:2px 7px;border-radius:5px;background:var(--t-amber);color:#92400E;margin-right:3px}
/* active discussions */
.disc{display:flex;flex-direction:column;gap:10px}
.disc-item{display:flex;gap:12px;align-items:flex-start;padding:12px 14px;border:1px solid var(--line);border-radius:11px;background:#FCFDFE}
.disc-item .badge{font-size:10px;font-weight:700;padding:3px 8px;border-radius:6px;text-transform:uppercase;letter-spacing:.05em;flex-shrink:0;margin-top:1px}
.badge.issue{background:var(--t-blue);color:var(--indigo)} .badge.pr{background:var(--t-mint);color:#0E7E72}
.disc-item .main{flex:1;min-width:0}
.disc-item .ttl{font-size:13px;font-weight:500;line-height:1.4}
.disc-item .meta{font-size:11px;color:var(--mute);margin-top:3px;font-family:var(--mono)}
.disc-item .cmt{font-family:var(--mono);font-weight:700;color:#B45309;font-size:13px;flex-shrink:0}
/* topic bars */
.tlist{display:flex;flex-direction:column;gap:11px}
.trow{display:flex;align-items:center;gap:12px}
.trow .tname{width:180px;font-size:13px;font-weight:500;font-family:var(--mono);color:var(--ink)}
.trow .tbar{flex:1;height:24px;background:#F1F5F9;border-radius:7px;overflow:hidden}
.trow .tbar i{display:block;height:100%;background:linear-gradient(90deg,#06B6D4,#22D3EE);border-radius:7px}
.trow .tval{font-family:var(--mono);font-weight:700;font-size:13px;width:48px;text-align:right}
/* scholars grid */
.sgrid{display:grid;grid-template-columns:repeat(auto-fill,minmax(200px,1fr));gap:12px}
.scard{border:1px solid var(--line);border-radius:11px;padding:13px 15px;background:#FCFDFE}
.scard .nm{font-family:var(--mono);font-weight:600;color:var(--ink);font-size:13px}
.scard .ct{font-size:12px;color:var(--mute);margin-top:3px}
.scard.org{border-left:3px solid var(--indigo)}
/* repos network mini-list */
.rlist{display:flex;flex-direction:column;gap:6px}
.rlist .rli{font-family:var(--mono);font-size:11px;color:var(--mute);padding:2px 0}
/* empty / loading / placeholder */
.empty{padding:50px 20px;text-align:center;color:var(--mute);font-size:14px}
.placeholder{margin-top:18px;padding:60px 20px;text-align:center;color:var(--mute);background:var(--card);
border:1px dashed var(--line);border-radius:16px;font-size:14px}
.placeholder b{color:var(--ink2)}
/* skeleton loading */
@keyframes shimmer{0%{background-position:-400px 0}100%{background-position:400px 0}}
.skel{background:linear-gradient(90deg,#F1F5F9 25%,#E6EBF2 50%,#F1F5F9 75%);background-size:800px 100%;animation:shimmer 1.5s infinite;border-radius:8px}
/* responsive */
@media(max-width:760px){
.wrap{padding:16px}
.summary{grid-template-columns:repeat(2,1fr)}
.sgrid{grid-template-columns:repeat(auto-fill,minmax(150px,1fr))}
.row .lbl{min-width:60px;font-size:11px}
}
</style>
</head>
<body>
<div class="wrap">
<div class="hdr">
<a class="back" href="index.html">← 仪表盘</a>
<div class="titles">
<h1>科研热点追踪</h1>
<div class="sub">输入领域/关键词,勾选追踪角度,扫描 GitLink 科研仓库 → 飙升项目 · 活跃讨论 · 主题热度 · 学者团队</div>
</div>
</div>
<!-- control panel -->
<div class="ctl">
<div class="row">
<span class="lbl">领域 / 关键词</span>
<label class="search-input"><span>🔍</span>
<input id="keyword" value="deep learning,机器学习" placeholder='输入研究领域或技术关键词,逗号分隔多个(如 "计算机视觉,NLP,transformer"'>
</label>
<button class="btn-run" id="runBtn">开始追踪 ▶</button>
</div>
<div class="row">
<span class="lbl">示例</span>
<div class="ex-chips" id="exChips">
<span class="ex">深度学习</span><span class="ex">computer vision</span>
<span class="ex">自然语言处理</span><span class="ex">强化学习</span>
<span class="ex">knowledge graph</span><span class="ex">联邦学习</span>
<span class="ex">LLM agent</span><span class="ex">自动驾驶</span>
</div>
</div>
<div class="row">
<span class="lbl">追踪角度</span>
<div class="angles" id="angles">
<span class="ang on" data-a="trending"><span class="dot"></span>🔥 飙升项目</span>
<span class="ang on" data-a="active"><span class="dot"></span>💬 活跃讨论</span>
<span class="ang on" data-a="topics"><span class="dot"></span>📊 热门主题</span>
<span class="ang on" data-a="scholars"><span class="dot"></span>👥 核心学者·团队</span>
</div>
<span class="lbl" style="margin-left:auto">仓库上限</span>
<select class="lim" id="lim">
<option>5</option><option>8</option><option selected>12</option><option>20</option><option>30</option>
</select>
</div>
<div class="status" id="status"><span>就绪 · 点"开始追踪"扫描 GitLink</span></div>
</div>
<!-- summary tiles (hidden until first run) -->
<div class="summary" id="summary" style="display:none"></div>
<!-- results -->
<div id="results">
<div class="placeholder">输入关键词并选择角度,点 <b>开始追踪</b> —— 服务器将通过 gitlink-cli 实时搜索并构建该领域的科研热点全景。</div>
</div>
</div>
<script>
/* ========== 工具函数 ========== */
function esc(s){return String(s||"").replace(/[&<>]/g,c=>({'&':'&amp;','<':'&lt;','>':'&gt;'}[c]));}
function token(){return localStorage.getItem("demo_token")||"";}
/* ========== angle 切换 + 示例芯片 ========== */
document.querySelectorAll(".ang").forEach(a=>a.onclick=()=>a.classList.toggle("on"));
document.querySelectorAll(".ex").forEach(e=>e.onclick=()=>{
document.getElementById("keyword").value=e.textContent.trim();
});
/* ========== 状态栏 ========== */
function setStatus(html,cls){
const s=document.getElementById("status");
s.innerHTML=html;
s.className="status"+(cls?" "+cls:"");
}
/* ========== 主运行 ========== */
async function run(){
const kw=document.getElementById("keyword").value.trim();
if(!kw){setStatus("请输入关键词","err");return;}
const angles=[...document.querySelectorAll(".ang.on")].map(a=>a.dataset.a);
if(!angles.length){setStatus("请至少选择一个追踪角度","err");return;}
const btn=document.getElementById("runBtn");
btn.disabled=true;
setStatus('<span class="spin"></span> 扫描中:搜索 "'+esc(kw)+'" → 拉取仓库详情 → 批量获取 Issue/PR 讨论…(约 3090s取决于仓库数和 API 响应速度)');
document.getElementById("results").innerHTML="";
document.getElementById("summary").style.display="none";
try{
const r=await fetch("/api/run",{
method:"POST",
headers:{"Content-Type":"application/json","X-Demo-Token":token()},
body:JSON.stringify({scenario:"hotspot",keyword:kw})
});
const j=await r.json();
if(!j.ok){
const errMsg=j.error||j.stdout||"未知错误";
setStatus("✗ 运行失败: "+esc(errMsg)+" · "+(j.duration||""),"err");
btn.disabled=false;
return;
}
// 解析 JSON
let obj=null;
try{
const raw=(j.artifacts&&j.artifacts.json)||(j.stdout||"");
obj=typeof raw==="string"?JSON.parse(raw):raw;
}catch(e){
setStatus("✗ 解析 JSON 产物失败 · 可能是 Python 脚本异常: "+(e.message||""),"err");
btn.disabled=false;
return;
}
if(!obj||typeof obj!=="object"){
setStatus("✗ 返回数据为空 · 请确认目标关键词在 GitLink 上有结果","err");
btn.disabled=false;
return;
}
const meta=obj.meta||{};
setStatus(
"✓ 完成 · "+(j.duration||"")+
" · 仓库 "+meta.repo_count+" 个"+
" · 讨论 "+meta.discussion_count+" 条"+
" · 主题 "+meta.topic_count+" 个"+
" · 学者 "+meta.scholar_count+" 位",
"ok"
);
// 汇总卡片
renderSummary(meta);
// 结果
render(angles,obj);
}catch(e){
setStatus("✗ 请求失败: "+esc(e.message||e),"err");
}finally{
btn.disabled=false;
}
}
/* ========== 汇总卡片 ========== */
function renderSummary(meta){
const el=document.getElementById("summary");
el.style.display="grid";
el.innerHTML=
tile("indigo","扫描仓库",meta.repo_count||0,"个")+
tile("coral","活跃讨论",meta.discussion_count||0,"条")+
tile("cyan","热门主题",meta.topic_count||0,"个")+
tile("mint","核心学者",meta.scholar_count||0,"位");
}
function tile(c,label,val,unit){
const colors={indigo:"var(--indigo)",coral:"var(--coral)",cyan:"#06B6D4",mint:"var(--mint)"};
const bg={indigo:"var(--t-blue)",coral:"var(--t-coral)",cyan:"#E0F7FA",mint:"var(--t-mint)"};
return `<div class="sum-tile" style="border-top:3px solid ${colors[c]||colors.indigo}">
<div class="sl">${label}</div><div class="sv">${val}</div><div class="ss">${unit}</div></div>`;
}
/* ========== 渲染各个角度 ========== */
function render(angles,o){
const box=document.getElementById("results");
let h="";
const maxScore=Math.max(1,...(o.trending_repos||[]).map(r=>r.score||0));
// --- 🔥 飙升项目 ---
if(angles.includes("trending")){
const tr=o.trending_repos||[];
h+=`<div class="sec">
<div class="sec-head">
<div class="ico ic-coral">🔥</div>
<div>
<h2>飙升 / 热门项目</h2>
<span class="eyebrow">Trending repos · 热度 = Star + Fork×2 + 近期更新加成</span>
</div>
<span class="cnt">${tr.length} 个</span>
</div>`;
if(tr.length){
h+=`<table class="tbl"><thead><tr>
<th>仓库</th><th>语言</th><th class="num">★ Star</th><th class="num">⑂ Fork</th><th class="num">日均★</th><th>热度</th><th>更新</th></tr></thead><tbody>`;
tr.forEach(r=>{
const kws=(r.matched_keywords||[]).map(k=>`<span class="kw-chip">${esc(k)}</span>`).join("");
h+=`<tr>
<td><div class="repo">${esc(r.repo)}</div>
${kws?`<div style="margin-top:3px">${kws}</div>`:""}
<div style="font-size:11px;color:var(--mute);margin-top:2px">${esc(r.description||"").slice(0,100)}</div></td>
<td>${r.language?`<span class="lang-pill">${esc(r.language)}</span>`:"—"}</td>
<td class="num">${r.stars}</td>
<td class="num">${r.forks}</td>
<td class="num">${r.velocity}</td>
<td>
<div class="scorebar"><i style="width:${Math.round(r.score/maxScore*100)}%"></i></div>
<div style="font-size:11px;color:var(--mute);margin-top:2px;font-family:var(--mono)">${r.score}</div>
</td>
<td style="font-family:var(--mono);font-size:12px;color:var(--ink2)">${esc(r.updated||"—")}</td>
</tr>`;
});
h+=`</tbody></table>`;
}else{
h+=`<div class="empty">未找到匹配仓库</div>`;
}
h+=`</div>`;
}
// --- 💬 活跃讨论 ---
if(angles.includes("active")){
const ac=o.active_discussions||[];
h+=`<div class="sec">
<div class="sec-head">
<div class="ico ic-amber">💬</div>
<div>
<h2>活跃讨论</h2>
<span class="eyebrow">Active issues / PRs · 按评论数排序 · 来自所有扫描仓库</span>
</div>
<span class="cnt">${ac.length} 条</span>
</div>`;
if(ac.length){
h+=`<div class="disc">`;
ac.slice(0,30).forEach(a=>{
h+=`<div class="disc-item">
<span class="badge ${a.type}">${a.type==="pr"?"PR":"Issue"}</span>
<div class="main">
<div class="ttl">${esc(a.title||"(无标题)")}</div>
<div class="meta">${esc(a.repo)} · #${esc(a.number||"")} · ${esc(a.state||"")}</div>
</div>
<span class="cmt">${a.comments} 💬</span>
</div>`;
});
h+=`</div>`;
}else{
h+=`<div class="empty">该领域暂无活跃讨论(所有仓库的开放 Issue/PR 评论数均为 0</div>`;
}
h+=`</div>`;
}
// --- 📊 热门主题 ---
if(angles.includes("topics")){
const tp=o.topic_heat||[];
const maxT=Math.max(1,...tp.map(t=>t.count));
h+=`<div class="sec">
<div class="sec-head">
<div class="ico ic-cyan">📊</div>
<div>
<h2>热门主题</h2>
<span class="eyebrow">Topic heat · 在仓库描述/README 中命中的主题词频</span>
</div>
<span class="cnt">${tp.length} 个</span>
</div>`;
if(tp.length){
h+=`<div class="tlist">`;
tp.forEach(t=>{
h+=`<div class="trow">
<span class="tname">${esc(t.topic)}</span>
<div class="tbar"><i style="width:${Math.round(t.count/maxT*100)}%"></i></div>
<span class="tval">${t.count}</span>
</div>`;
});
h+=`</div>`;
}else{
h+=`<div class="empty">未识别到明确主题(仓库描述/README 不包含已知科研关键词)</div>`;
}
h+=`</div>`;
}
// --- 👥 核心学者·团队 ---
if(angles.includes("scholars")){
const sc=o.core_scholars||[];
const tm=o.core_teams||[];
h+=`<div class="sec">
<div class="sec-head">
<div class="ico ic-mint">👥</div>
<div>
<h2>核心学者 · 团队</h2>
<span class="eyebrow">Key scholars & teams · 按关联仓库数排列</span>
</div>
<span class="cnt">学者 ${(sc||[]).length} · 团队 ${(tm||[]).length}</span>
</div>`;
// 学者
if(sc.length){
h+=`<div style="margin-bottom:14px"><div style="font-size:12px;font-weight:600;color:var(--ink2);margin-bottom:8px">🧑 学者</div>`;
h+=`<div class="sgrid">`;
sc.slice(0,12).forEach(s=>{
const repoNames=(s.repos||[]).slice(0,5);
h+=`<div class="scard">
<div class="nm">${esc(s.login)}</div>
<div class="ct">关联 ${s.repo_count} 个仓库</div>
${repoNames.length?`<div class="rlist">${repoNames.map(rn=>`<span class="rli">${esc(rn)}</span>`).join("")}</div>`:""}
</div>`;
});
h+=`</div></div>`;
}
// 团队
if(tm.length){
h+=`<div><div style="font-size:12px;font-weight:600;color:var(--ink2);margin-bottom:8px">🏛 活跃组织/团队</div>`;
h+=`<div class="sgrid">`;
tm.slice(0,8).forEach(t=>{
h+=`<div class="scard org">
<div class="nm">${esc(t.login)}</div>
<div class="ct">${t.repo_count} 个仓库</div>
</div>`;
});
h+=`</div></div>`;
}
if(!sc.length&&!tm.length){
h+=`<div class="empty">未识别到学者/团队(仓库无贡献者数据)</div>`;
}
h+=`</div>`;
}
box.innerHTML=h||'<div class="empty" style="padding:70px 20px">无匹配角度的数据 —— 请选择至少一个追踪角度</div>';
}
/* ========== 事件绑定 ========== */
document.getElementById("runBtn").onclick=run;
document.getElementById("keyword").addEventListener("keydown",e=>{if(e.key==="Enter")run();});
/* ========== 页面加载时尝试载入上次的 hotspot 结果 ========== */
(async function init(){
try{
const r=await fetch("/api/result/hotspot",{headers:{"X-Demo-Token":token()}});
const j=await r.json();
if(j.ok&&j.artifacts&&j.artifacts.json){
const obj=JSON.parse(j.artifacts.json);
if(obj&&obj.trending_repos){
const angles=[...document.querySelectorAll(".ang.on")].map(a=>a.dataset.a);
const meta=obj.meta||{};
renderSummary(meta);
render(angles,obj);
setStatus("✓ 已载入上次扫描 · 点"开始追踪"用新关键词重新扫描","ok");
}
}
}catch(e){}
})();
</script>
</body>
</html>

View File

@ -0,0 +1,594 @@
<!DOCTYPE html>
<html lang="zh-CN">
<head>
<meta charset="UTF-8">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>GitLink Research Atlas · 科研代码图谱智能体</title>
<link rel="preconnect" href="https://fonts.googleapis.com">
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
<link href="https://fonts.googleapis.com/css2?family=Cormorant+Garamond:wght@500;600;700&family=Inter:wght@400;500;600;700&family=JetBrains+Mono:wght@500;700&family=Noto+Serif+SC:wght@600;700&family=Noto+Sans+SC:wght@400;500;700&display=swap" rel="stylesheet">
<script src="https://cdn.jsdelivr.net/npm/mermaid@10/dist/mermaid.min.js"></script>
<style>
:root{
--bg:#F7F9FC; --card:#FFFFFF; --ink:#1E2A44; --ink2:#475569; --mute:#94A3B8;
--line:#E6EBF2; --indigo:#4F6BED; --coral:#F26B5E; --mint:#2EC4B6;
--t-blue:#E8EEFF; --t-coral:#FFEDEA; --t-mint:#E6F7F4; --t-amber:#FEF3C7;
--drawer:#111A2E;
--serif:"Cormorant Garamond","Noto Serif SC",Georgia,serif;
--sans:"Inter","Noto Sans SC",-apple-system,"PingFang SC","Microsoft YaHei",sans-serif;
--mono:"JetBrains Mono",ui-monospace,SFMono-Regular,Menlo,Consolas,monospace;
--shadow:0 1px 2px rgba(30,42,68,.04),0 6px 18px rgba(30,42,68,.06);
--shadow-sm:0 1px 2px rgba(30,42,68,.06);
}
*{box-sizing:border-box;margin:0;padding:0}
html,body{height:100%;margin:0;overflow:hidden}
body{font-family:var(--sans);color:var(--ink);background:var(--bg);
background-image:radial-gradient(1200px 480px at 80% -8%,#EAF0FF 0%,rgba(234,240,255,0) 60%),
radial-gradient(900px 420px at 0% 0%,#FFF1EE 0%,rgba(255,241,238,0) 55%);
-webkit-font-smoothing:antialiased}
.app{display:flex;flex-direction:column;height:100vh;overflow:hidden}
/* ========== HEADER ========== */
header.bar{
display:flex;align-items:center;gap:18px;padding:0 22px;
min-height:62px;height:auto;
background:rgba(255,255,255,.86);backdrop-filter:blur(8px);
border-bottom:1px solid var(--line);position:relative;z-index:5;
overflow:visible;flex-shrink:0;
}
.brand{display:flex;align-items:center;gap:10px;min-width:0;flex-shrink:0}
.brand .mark{width:30px;height:30px;border-radius:9px;background:linear-gradient(135deg,var(--indigo),#7C8CF5 60%,var(--coral));
display:grid;place-items:center;color:#fff;font-family:var(--serif);font-weight:700;font-size:18px;box-shadow:var(--shadow-sm)}
.brand .t1{font-family:var(--serif);font-weight:700;font-size:20px;line-height:1.1;letter-spacing:.2px}
.brand .t2{font-size:10px;color:var(--mute);letter-spacing:.12em;text-transform:uppercase;line-height:1.2}
.search{display:flex;align-items:center;gap:8px;margin-left:8px;flex:1;max-width:640px}
.field{display:flex;align-items:center;gap:7px;background:var(--card);border:1px solid var(--line);
border-radius:10px;padding:7px 11px;box-shadow:var(--shadow-sm);flex:1}
.field .k{font-family:var(--mono);font-size:11px;color:var(--mute)}
.field input{border:0;outline:0;font-family:var(--mono);font-size:13px;color:var(--ink);width:100%;background:transparent}
.chip{display:inline-flex;align-items:center;gap:6px;font-size:12px;font-weight:600;padding:6px 11px;border-radius:999px;white-space:nowrap}
.chip.agent{background:var(--t-mint);color:#0E7E72;border:1px solid #B6EBE3}
.chip.agent::before{content:"";width:7px;height:7px;border-radius:50%;background:var(--mint);box-shadow:0 0 0 3px rgba(46,196,182,.18);animation:pulse 1.8s infinite}
.btn-login{font-size:12px;font-weight:600;color:#fff;background:var(--indigo);border:0;border-radius:10px;padding:8px 15px;cursor:pointer;box-shadow:var(--shadow-sm)}
.btn-hot{font-size:12px;font-weight:600;color:#fff;background:linear-gradient(135deg,var(--coral),#FF8A7A);text-decoration:none;border:0;border-radius:10px;padding:8px 14px;cursor:pointer;box-shadow:var(--shadow-sm);display:inline-flex;align-items:center;gap:5px;white-space:nowrap}
.btn-hot:hover{filter:brightness(1.05)}
@keyframes pulse{50%{opacity:.4}}
/* ========== MAIN GRID ========== */
main.grid{
display:grid;grid-template-columns:262px 1fr 332px;
gap:16px;padding:16px;flex:1;min-height:0;overflow:hidden;
}
aside.left,aside.right{min-height:0;overflow:auto;display:flex;flex-direction:column;gap:12px;padding-right:2px}
section.center{min-height:0;display:flex;flex-direction:column;gap:14px;overflow:hidden}
.panel{background:var(--card);border:1px solid var(--line);border-radius:14px;box-shadow:var(--shadow)}
.panel .ph{padding:14px 16px 6px}
.panel .ph h3{margin:0;font-size:13px;font-weight:700;letter-spacing:.02em}
.panel .ph .sub{font-size:11px;color:var(--mute);margin-top:2px}
.eyebrow{font-family:var(--mono);font-size:10px;letter-spacing:.16em;text-transform:uppercase;color:var(--mute)}
/* ========== SCENARIO CARDS ========== */
.scn{display:flex;gap:11px;align-items:flex-start;padding:11px 12px;border:1px solid var(--line);border-radius:12px;
background:var(--card);cursor:pointer;transition:.15s;position:relative;box-shadow:var(--shadow-sm)}
.scn:hover{border-color:#C7D2FE;transform:translateY(-1px)}
.scn.active{border-color:var(--indigo);background:linear-gradient(180deg,#F5F8FF,#fff)}
.scn.active::after{content:"";position:absolute;top:10px;right:10px;width:7px;height:7px;border-radius:50%;background:var(--coral)}
.scn.running{pointer-events:none;opacity:.7}
.scn .ico{width:30px;height:30px;border-radius:9px;display:grid;place-items:center;flex:0 0 auto;font-size:15px}
.scn .t{min-width:0}
.scn .t b{font-size:13px;font-weight:600;display:block}
.scn .t small{font-size:11px;color:var(--mute);line-height:1.35;display:block;margin-top:1px}
.tint-blue .ico{background:var(--t-blue);color:var(--indigo)}
.tint-coral .ico{background:var(--t-coral);color:var(--coral)}
.tint-mint .ico{background:var(--t-mint);color:#0E7E72}
.tint-amber .ico{background:var(--t-amber);color:#B45309}
/* ========== CENTER HERO / GRAPH ========== */
.hero{flex:1;min-height:0;display:flex;flex-direction:column;overflow:hidden}
.hero .ph{display:flex;align-items:center;justify-content:space-between;flex-shrink:0}
.hero .ph .live{font-family:var(--mono);font-size:11px;color:var(--indigo);background:var(--t-blue);padding:4px 9px;border-radius:999px}
.graph-wrap{flex:1;min-height:0;position:relative;margin:0 4px 4px;border-radius:10px;
background:radial-gradient(420px 320px at 50% 42%,#F2F6FF 0%,rgba(247,249,252,0) 70%);overflow:hidden}
svg.graph{width:100%;height:100%;display:block}
.g-edge{stroke:#9DB4F0;stroke-opacity:.55;stroke-width:1}
.g-node circle{stroke:#fff;stroke-width:2;transition:.2s}
.g-node:hover circle{filter:brightness(1.08)}
.g-node text{font-family:var(--sans);font-size:10px;font-weight:600;fill:var(--ink);pointer-events:none;text-anchor:middle}
.glow{filter:drop-shadow(0 0 6px rgba(242,107,94,.55))}
/* hero-overlay: ONLY visible during loading/empty, hidden after graph renders */
.hero-overlay{position:absolute;inset:0;display:flex;align-items:center;justify-content:center;flex-direction:column;gap:8px;pointer-events:none;
transition:opacity .3s}
.hero-overlay.hidden{opacity:0;pointer-events:none}
.hero-overlay .hint{font-family:var(--mono);font-size:12px;color:var(--mute);background:rgba(255,255,255,.8);padding:6px 12px;border-radius:999px}
.spin{width:26px;height:26px;border-radius:50%;border:3px solid var(--t-blue);border-top-color:var(--indigo);animation:rot .8s linear infinite}
@keyframes rot{to{transform:rotate(360deg)}}
.center-extra{flex:0 0 auto;max-height:46%;display:flex;flex-direction:column;gap:10px;overflow:auto}
.center-extra:empty{display:none}
iframe.plot{width:100%;height:420px;border:0;border-radius:10px;background:#fff}
.mmd-box{background:#fff;padding:8px;border-radius:10px;overflow:auto;max-height:420px}
/* Graph legend bar */
.graph-legend{display:flex;align-items:center;gap:14px;padding:6px 16px 10px;flex-wrap:wrap;flex-shrink:0}
.graph-legend .leg-item{display:flex;align-items:center;gap:5px;font-size:11px;color:var(--ink2)}
.graph-legend .leg-dot{width:10px;height:10px;border-radius:50%;flex-shrink:0}
.graph-legend .leg-note{font-size:11px;color:var(--mute);margin-left:auto}
/* ========== RIGHT COLUMN ========== */
.metrics{display:grid;grid-template-columns:1fr 1fr;gap:10px;padding:6px 14px 14px}
.tile{border:1px solid var(--line);border-radius:12px;padding:12px;background:#FCFDFE;box-shadow:var(--shadow-sm)}
.tile .lab{display:flex;align-items:center;gap:6px;font-size:11px;color:var(--ink2);font-weight:500}
.tile .lab .dot{width:8px;height:8px;border-radius:50%}
.tile .val{font-family:var(--mono);font-weight:700;font-size:24px;line-height:1.1;margin-top:6px;color:var(--ink)}
.tile .of{font-size:11px;color:var(--mute);margin-top:3px}
.d-indigo{background:var(--indigo)} .d-mint{background:var(--mint)} .d-coral{background:var(--coral)} .d-amber{background:#F59E0B}
.findings{padding:4px 14px 14px;display:flex;flex-direction:column;gap:9px;max-height:330px;overflow:auto}
.find{border-left:3px solid var(--indigo);padding:8px 11px;background:#F8FAFF;border-radius:0 9px 9px 0}
.find.critical{border-color:var(--coral);background:#FFF5F3}
.find.warning{border-color:#F59E0B;background:#FFFBEB}
.find.success{border-color:var(--mint);background:#F0FBF9}
.find .ft{font-size:11px;font-weight:700;color:var(--ink2);display:flex;justify-content:space-between;gap:8px}
.find .fb{font-size:12px;color:var(--ink);margin-top:3px;line-height:1.45}
.find .fs{font-size:11px;color:var(--mute);margin-top:3px}
.bar{height:6px;border-radius:4px;background:var(--line);overflow:hidden;margin-top:6px}
.bar > i{display:block;height:100%;background:linear-gradient(90deg,var(--indigo),#7C8CF5)}
.empty{font-size:12px;color:var(--mute);padding:18px;text-align:center}
.loading-brief{font-size:12px;color:var(--mute);padding:18px;text-align:center;display:flex;flex-direction:column;align-items:center;gap:8px}
.loading-brief .spin{width:20px;height:20px;border-width:2px}
/* CTA button to view full result */
.view-result-btn{display:block;margin:8px 14px 14px;padding:10px 18px;border:0;border-radius:12px;
background:linear-gradient(135deg,var(--indigo),#7C8CF5);color:#fff;font-family:var(--sans);
font-size:13px;font-weight:600;cursor:pointer;text-align:center;box-shadow:var(--shadow);
transition:.15s}
.view-result-btn:hover{transform:translateY(-1px);box-shadow:0 4px 14px rgba(79,107,237,.25)}
/* ========== COMMAND DRAWER ========== */
footer.drawer{display:flex;align-items:center;gap:14px;padding:0 18px;background:var(--drawer);color:#CBD5E1;font-family:var(--mono);font-size:12px;
flex-shrink:0;min-height:52px;height:52px}
.dots{display:flex;gap:6px}
.dots i{width:11px;height:11px;border-radius:50%;display:block}
.drawer .prompt{color:var(--mint)}
.drawer .cmd{flex:1;white-space:nowrap;overflow:hidden;text-overflow:ellipsis;color:#E2E8F0}
.drawer .st{color:var(--mute);font-size:11px;white-space:nowrap}
.drawer .st.running{color:var(--indigo);animation:pulse 1s infinite}
/* ========== SCROLLBARS ========== */
aside::-webkit-scrollbar,.findings::-webkit-scrollbar,.center-extra::-webkit-scrollbar{width:8px;height:8px}
::-webkit-scrollbar-thumb{background:#D7DEEA;border-radius:8px}
::-webkit-scrollbar-track{background:transparent}
/* ========== RESPONSIVE ========== */
@media (max-width:1180px){
main.grid{grid-template-columns:230px 1fr;grid-template-rows:auto auto}
aside.right{grid-column:1 / span 2;flex-direction:row;flex-wrap:wrap}
aside.right .panel{flex:1;min-width:280px}
}
@media (max-width:760px){
main.grid{grid-template-columns:1fr;overflow:auto}
aside.left,aside.right{grid-column:auto}
.search{display:none}
}
</style>
</head>
<body>
<div class="app">
<!-- ============ TOP BAR ============ -->
<header class="bar">
<div class="brand">
<div class="mark">G</div>
<div>
<div class="t1">GitLink Research Atlas</div>
<div class="t2">科研代码图谱智能体</div>
</div>
</div>
<div class="search">
<label class="field"><span class="k">repo</span>
<input id="owner" value="mindspore-Ecosystem" placeholder="owner">
<span class="k" style="color:#CBD5E1">/</span>
<input id="repo" value="mindspore" placeholder="repo" style="max-width:150px">
</label>
<label class="field"><span class="k">keyword</span>
<input id="keyword" value="deep learning,机器学习" placeholder="S2 用,逗号分隔">
</label>
</div>
<a class="btn-hot" href="hotspot.html">🔥 热点追踪</a>
<span class="chip agent">AI Agent online</span>
<button class="btn-login" id="loginBtn">Login</button>
</header>
<!-- ============ MAIN ============ -->
<main class="grid">
<!-- LEFT: scenarios -->
<aside class="left">
<div class="panel">
<div class="ph"><div class="eyebrow">Research Scenarios</div>
<h3>科研场景</h3><div class="sub">点选任一场景,服务器终端即运行 gitlink-cli + Python 算法</div>
</div>
<div id="scenarios" style="padding:8px 12px 14px;display:flex;flex-direction:column;gap:9px"></div>
</div>
</aside>
<!-- CENTER: graph / viz -->
<section class="center">
<div class="panel hero">
<div class="ph">
<div><div class="eyebrow">Knowledge Graph</div><h3 id="heroTitle">科研知识图谱</h3></div>
<span class="live" id="heroLive" style="display:none">&#9679; live build</span>
</div>
<div class="graph-wrap">
<svg class="graph" id="graph" preserveAspectRatio="xMidYMid meet"></svg>
<div class="hero-overlay hidden" id="heroOverlay">
<div class="hint">选择左侧科研场景以启动分析</div>
</div>
</div>
</div>
<div class="graph-legend" id="graphLegend"></div>
<div class="center-extra" id="centerExtra"></div>
</section>
<!-- RIGHT: metrics + report -->
<aside class="right">
<div class="panel">
<div class="ph"><div class="eyebrow">Live Metrics</div><h3>科研指标</h3></div>
<div class="metrics" id="metrics"></div>
</div>
<div class="panel" style="flex:1;min-height:0;display:flex;flex-direction:column">
<div class="ph"><div class="eyebrow">AI Research Brief</div><h3 id="briefTitle">研究简报</h3></div>
<div class="findings" id="findings"><div class="empty">运行场景后,这里展示创新点 / 风险预警 / 协作推荐等结构化结论</div></div>
<button class="view-result-btn" id="viewResultBtn" style="display:none" target="_blank">查看完整结果 &#8594;</button>
</div>
</aside>
</main>
<!-- ============ DRAWER ============ -->
<footer class="drawer">
<div class="dots"><i style="background:#FF5F57"></i><i style="background:#FEBC2E"></i><i style="background:#28C840"></i></div>
<span class="prompt">atlas &#10095;</span>
<span class="cmd" id="cmd">gitlink-cli research --repo mindspore-Ecosystem/mindspore --agent atlas</span>
<span class="st" id="cmdStatus">ready</span>
</footer>
</div>
<script>
mermaid.initialize({startOnLoad:false, theme:"base", securityLevel:"loose",
themeVariables:{primaryColor:"#E8EEFF",primaryTextColor:"#1E2A44",primaryBorderColor:"#4F6BED",lineColor:"#9DB4F0",fontSize:"12px"}});
let SCENARIOS=[], ACTIVE=null, LAST_RESULT_KEY=null;
/* ---------- node color palette ---------- */
const NODE_COLOR = {
repo:"#4F6BED", scholar:"#3B82F6", team:"#6366F1", topic:"#06B6D4", method:"#6366F1",
paper:"#8B5CF6", dataset:"#2EC4B6", model:"#4F6BED", experiment:"#F26B5E",
reproduce:"#2EC4B6", license:"#64748B", trend:"#F26B5E", issue:"#F59E0B", pr:"#10B981",
default:"#94A3B8"
};
const NODE_LABEL = {
repo:"仓库", scholar:"学者", topic:"主题", paper:"论文", dataset:"数据集",
model:"模型", experiment:"实验", reproduce:"复现", license:"许可证",
trend:"趋势", issue:"Issue", pr:"PR"
};
/* Default decorative graph — clean, ~12 nodes showing the research world */
const DEMO = {
nodes:[
{id:"repo",type:"repo",label:"GitLink Repo",imp:true},
{id:"scholar",type:"scholar",label:"Scholar"},
{id:"topic",type:"topic",label:"Topic"},
{id:"paper",type:"paper",label:"Paper"},
{id:"dataset",type:"dataset",label:"Dataset"},
{id:"model",type:"model",label:"Model"},
{id:"exp",type:"experiment",label:"Experiment",imp:true},
{id:"repr",type:"reproduce",label:"Reproduce"},
{id:"lic",type:"license",label:"License"},
{id:"trend",type:"trend",label:"Trend"},
{id:"issue",type:"issue",label:"Issue Signals"},
{id:"pr",type:"pr",label:"Pull Request"}
],
edges:[["repo","scholar"],["repo","topic"],["repo","paper"],["repo","dataset"],
["repo","model"],["repo","exp"],["repo","repr"],["repo","lic"],
["scholar","paper"],["topic","trend"],["exp","repr"],["issue","pr"],["pr","repo"],["model","dataset"]]
};
/* ---------- build legend HTML ---------- */
function buildLegend(){
const el=document.getElementById("graphLegend");
const types=new Set();
// gather types from current graph data
const src=ACTIVE&&ACTIVE._graphData?ACTIVE._graphData:DEMO;
(src.nodes||[]).forEach(n=>types.add(n.type||"default"));
let h="";
for(const t of [...types].sort()){
h+=`<span class="leg-item"><span class="leg-dot" style="background:${NODE_COLOR[t]||NODE_COLOR.default}"></span>${NODE_LABEL[t]||t}</span>`;
}
h+=`<span class="leg-note">示例:科研仓库与学者/主题/数据集的关联 — 选择 S2 生成真实领域图谱</span>`;
el.innerHTML=h;
}
/* ---------- force-directed layout ---------- */
function forceLayout(nodes, edges, W, H, iters){
const cx=W/2, cy=H/2;
nodes.forEach(n=>{ if(n.x==null){n.x=cx+(Math.random()-.5)*W*.5; n.y=cy+(Math.random()-.5)*H*.5;} });
for(let it=0; it<iters; it++){
for(const n of nodes){ n._fx=0; n._fy=0; }
for(let i=0;i<nodes.length;i++){
for(let j=i+1;j<nodes.length;j++){
let dx=nodes[i].x-nodes[j].x, dy=nodes[i].y-nodes[j].y;
let d2=dx*dx+dy*dy+0.02, d=Math.sqrt(d2), f=2600/d2;
let fx=dx/d*f, fy=dy/d*f;
nodes[i]._fx+=fx; nodes[i]._fy+=fy; nodes[j]._fx-=fx; nodes[j]._fy-=fy;
}
}
for(const e of edges){
const a=nodes.find(n=>n.id===e[0]), b=nodes.find(n=>n.id===e[1]); if(!a||!b) continue;
let dx=b.x-a.x, dy=b.y-a.y, d=Math.sqrt(dx*dx+dy*dy)+.02, L=a.imp||b.imp?120:78, k=.04;
let f=(d-L)*k, fx=dx/d*f, fy=dy/d*f;
a._fx+=fx; a._fy+=fy; b._fx-=fx; b._fy-=fy;
}
for(const n of nodes){
n._fx+=(cx-n.x)*.012; n._fy+=(cy-n.y)*.012;
n.x+=Math.max(-14,Math.min(14,n._fx)); n.y+=Math.max(-14,Math.min(14,n._fy));
n.x=Math.max(40,Math.min(W-40,n.x)); n.y=Math.max(34,Math.min(H-30,n.y));
}
}
}
function drawGraph(data){
const svg=document.getElementById("graph");
const W=svg.clientWidth||760, H=svg.clientHeight||460;
svg.setAttribute("viewBox",`0 0 ${W} ${H}`);
let nodes=(data.nodes||[]).map(n=>Object.assign({},n));
let rawEdges=(data.edges||[]);
// normalize edges to [source,target] pairs
let edges=rawEdges.map(e=>Array.isArray(e)?e:[e.source||e.from,e.target||e.to]);
// cap to ~40 for performance
if(nodes.length>40){
const deg=new Map(nodes.map(n=>[n.id,0]));
edges.forEach(e=>{deg.set(e[0],(deg.get(e[0])||0)+1);deg.set(e[1],(deg.get(e[1])||0)+1);});
nodes.sort((a,b)=>(deg.get(b.id)||0)-(deg.get(a.id)||0));
const keep=new Set(nodes.slice(0,40).map(n=>n.id));
nodes=nodes.filter(n=>keep.has(n.id));
edges=edges.filter(e=>keep.has(e[0])&&keep.has(e[1]));
}
forceLayout(nodes, edges, W, H, nodes.length>20?220:320);
let s=`<defs>
<filter id="glow"><feGaussianBlur stdDeviation="4" result="blur"/><feMerge><feMergeNode in="blur"/><feMergeNode in="SourceGraphic"/></feMerge></filter>
</defs>`;
for(const e of edges){
const a=nodes.find(n=>n.id===e[0]), b=nodes.find(n=>n.id===e[1]); if(!a||!b) continue;
s+=`<line class="g-edge" x1="${a.x.toFixed(1)}" y1="${a.y.toFixed(1)}" x2="${b.x.toFixed(1)}" y2="${b.y.toFixed(1)}"/>`;
}
for(const n of nodes){
const col=NODE_COLOR[n.type]||NODE_COLOR.default;
const r=n.imp?11:7;
const glowAttr=n.imp?` filter="url(#glow)"`:"";
s+=`<g class="g-node" transform="translate(${n.x.toFixed(1)},${n.y.toFixed(1)})">
<circle r="${r}" fill="${col}"${glowAttr}/><text y="${r+12}">${esc(n.label||n.id).slice(0,16)}</text></g>`;
}
svg.innerHTML=s;
// BUG FIX #1: after graph renders, immediately hide the overlay
hideOverlay();
buildLegend();
}
/* ---------- overlay helpers ---------- */
const overlay=document.getElementById("heroOverlay");
function showOverlay(html){
overlay.innerHTML=html||'<div class="hint">选择左侧科研场景以启动分析</div>';
overlay.classList.remove("hidden");
}
function showLoading(msg){
overlay.innerHTML=`<div class="spin"></div><div class="hint">${esc(msg||"加载中...")}</div>`;
overlay.classList.remove("hidden");
}
function hideOverlay(){
overlay.classList.add("hidden");
}
/* ---------- init scenarios ---------- */
async function init(){
try{
const r=await fetch("/api/scenarios",{headers:{"X-Demo-Token":token()}});
const j=await r.json(); SCENARIOS=j.scenarios||[];
}catch(e){ SCENARIOS=[]; }
const tints=["tint-blue","tint-coral","tint-mint","tint-amber","tint-blue","tint-mint"];
const icons=["","Ⅱ","Ⅲ","Ⅳ","","Ⅵ"];
const box=document.getElementById("scenarios");
box.innerHTML=SCENARIOS.map((s,i)=>`
<div class="scn ${tints[i%tints.length]}" data-key="${s.key}">
<div class="ico">${icons[i]||"•"}</div>
<div class="t"><b>${esc(s.label)}</b><small>${esc(s.desc)}<br><span style="color:#A5B4CF">需: ${(s.needs||[]).join(" · ")}</span></small></div>
</div>`).join("");
box.querySelectorAll(".scn").forEach(el=>el.onclick=()=>run(el.dataset.key, el));
// Draw default graph
drawGraph(DEMO);
setMetrics(defaultMetrics());
buildLegend();
// Login button
document.getElementById("loginBtn").onclick=()=>{
const t=prompt("粘贴演示 token留空表示无鉴权", localStorage.getItem("demo_token")||"");
if(t!=null){ localStorage.setItem("demo_token",t.trim()); alert(t.trim()?"已保存":"已清除"); }
};
// View result button
document.getElementById("viewResultBtn").onclick=()=>{
if(LAST_RESULT_KEY) window.open("result.html?scenario="+LAST_RESULT_KEY,"_blank");
};
}
/* ---------- run scenario ---------- */
async function run(key, el){
const s=SCENARIOS.find(x=>x.key===key); if(!s) return;
ACTIVE=s;
// Mark card active + running
document.querySelectorAll(".scn").forEach(e=>{e.classList.remove("active");e.classList.remove("running");});
if(el) el.classList.add("active","running");
document.getElementById("heroTitle").textContent=s.label;
// BUG FIX #3: immediate loading feedback everywhere
showLoading("服务器终端执行中:"+s.script+" …");
document.getElementById("heroLive").style.display="none";
document.getElementById("cmd").textContent=cmdText(s);
document.getElementById("cmdStatus").textContent="running…";
document.getElementById("cmdStatus").className="st running";
document.getElementById("centerExtra").innerHTML="";
document.getElementById("viewResultBtn").style.display="none";
LAST_RESULT_KEY=null;
// BUG FIX #3: set brief + metrics to loading state
setFindings('<div class="loading-brief"><div class="spin"></div><div>正在分析…</div></div>');
setMetrics(tile("indigo","status","…","计算中")+tile("mint","status","…","")+tile("coral","status","…","")+tile("amber","status","…",""));
const body={scenario:key, owner:val("owner"), repo:val("repo"), keyword:val("keyword")};
try{
const r=await fetch("/api/run",{method:"POST",headers:{"Content-Type":"application/json","X-Demo-Token":token()},body:JSON.stringify(body)});
const j=await r.json();
// Clear running state
document.querySelectorAll(".scn").forEach(e=>e.classList.remove("running"));
document.getElementById("cmdStatus").textContent=(j.ok?"✓ done":"✗ failed")+" · "+(j.duration||"");
document.getElementById("cmdStatus").className="st";
if(!j.ok){
document.getElementById("heroOverlay").innerHTML='<div class="hint" style="color:#F26B5E">运行失败:'+esc(j.error||"未知错误")+'</div>';
document.getElementById("heroOverlay").classList.remove("hidden");
return;
}
renderResult(s, j);
}catch(e){
document.querySelectorAll(".scn").forEach(e=>e.classList.remove("running"));
document.getElementById("cmdStatus").textContent="✗ 请求失败";
document.getElementById("cmdStatus").className="st";
document.getElementById("heroOverlay").innerHTML='<div class="hint" style="color:#F26B5E">请求失败:'+esc(e)+'</div>';
document.getElementById("heroOverlay").classList.remove("hidden");
}
}
function cmdText(s){
const o=val("owner"), r=val("repo"), k=val("keyword");
if(s.key==="s2") return `gitlink-cli research --graph -k "${k}"`;
return `gitlink-cli research ${s.script.replace(".py","")} --repo ${o}/${r}`;
}
function val(id){return document.getElementById(id).value.trim();}
function token(){return localStorage.getItem("demo_token")||"";}
/* ---------- render result ---------- */
function parseJson(j){
if(!j.artifacts||!j.artifacts.json) return null;
try{return JSON.parse(j.artifacts.json);}catch(e){return null;}
}
function renderResult(s, j){
const obj=parseJson(j);
const extra=document.getElementById("centerExtra");
// S2: real knowledge graph
if(obj && obj.nodes && obj.edges){
const graphData={nodes:obj.nodes.map(n=>({id:n.id,type:n.type||"default",label:n.label||n.id,imp:false})), edges:obj.edges.map(e=>[e.source||e.from,e.target||e.to])};
drawGraph(graphData);
ACTIVE._graphData=graphData;
document.getElementById("heroLive").style.display="";
document.getElementById("heroLive").textContent="● live — "+(obj.nodes.length)+" nodes, "+(obj.edges.length)+" edges";
}
// S6: plotly iframe
else if(j.artifacts && j.artifacts.html){
const blob=URL.createObjectURL(new Blob([j.artifacts.html],{type:"text/html"}));
extra.innerHTML='<div class="panel" style="padding:10px"><div class="eyebrow" style="padding:4px 6px">S6 Visualization · plotly</div>'+
'<iframe class="plot" src="'+blob+'"></iframe></div>';
// restore demo graph if no real graph
drawGraph(DEMO);
ACTIVE._graphData=null;
}
// S1/S4: mermaid
else if(j.artifacts && j.artifacts.mermaid){
const id="mmd_"+Date.now();
extra.innerHTML='<div class="panel" style="padding:10px"><div class="eyebrow" style="padding:4px 6px">关系图 · Mermaid</div><div class="mmd-box"><div class="mermaid" id="'+id+'">'+j.artifacts.mermaid.replace(/```mermaid|```/g,"")+'</div></div></div>';
setTimeout(()=>mermaid.run({nodes:["#"+id]}).catch(()=>{}),30);
drawGraph(DEMO);
ACTIVE._graphData=null;
}
// S3/S5 or others: keep demo graph, hide overlay
else{
drawGraph(DEMO);
ACTIVE._graphData=null;
}
// Metrics + findings
if(obj){
setMetrics(metricsFor(s.key, obj));
setFindings(findingsFor(s.key, obj));
} else {
setMetrics(defaultMetrics());
setFindings('<div class="empty">'+(j.stdout?esc(j.stdout.slice(-400)):"(无可解析 JSON 产物)")+'</div>');
}
// Show "view full result" CTA
LAST_RESULT_KEY=s.key;
const btn=document.getElementById("viewResultBtn");
btn.style.display="block";
btn.textContent="查看完整结果 → "+s.label;
}
/* ---------- metrics per scenario ---------- */
function tile(c,label,val,of){return `<div class="tile"><div class="lab"><span class="dot d-${c}"></span>${label}</div><div class="val">${val}</div><div class="of">${of||""}</div></div>`;}
function defaultMetrics(){
return tile("indigo","commits","1,248","+18% this month")+
tile("mint","PR merge","72.6%","healthy collab")+
tile("indigo","contributors","36","8 core scholars")+
tile("coral","compliance","88","示例数据 · 待运行");
}
function metricsFor(key,o){
const M=o.meta||{};
switch(key){
case "s1": return tile("indigo","commits",M.commit_count||0,"采样提交")+tile("mint","merged PR",M.merged_pr_count||0,"已合并")+tile("amber","docs",M.doc_count||0,"文档文件")+tile("coral","创新点",(o.innovation_points||[]).length,"高影响合并");
case "s2": return tile("indigo","repos",M.repo_count||(o.nodes||[]).length,"仓库节点")+tile("coral","scholars",M.scholar_count||0,"学者")+tile("mint","topics",M.topic_count||0,"主题方向")+tile("amber","edges",M.edge_count||(o.edges||[]).length,"关系边");
case "s3": return tile("mint","repro",(o.repro_score||0)+"/10","复现性")+tile("indigo","compliance",(o.compliance_score||0)+"/10","合规")+tile("coral","risks",(o.risks||o.risk_items||[]).length,"风险项")+tile("amber","license",esc(o.license?(typeof o.license==="string"?o.license:(o.license.id||o.license.name||o.license.spdx_id||"—")):(o.license_id||"—")),"识别许可证");
case "s4": return tile("indigo","gap topics",(o.gap_topics||[]).length,"缺口主题")+tile("mint","candidates",(o.candidates||[]).length,"推荐候选")+tile("coral","top score",(o.candidates&&o.candidates[0]?o.candidates[0].score:0),"最高匹配")+tile("amber","pool",o.meta&&o.meta.pool_size||0,"候选池");
case "s5": {const tw=(o.week_stats&&o.week_stats.this_week)||{}; return tile("indigo","commits/周",tw.commits||0,"本周")+tile("coral","risks",(o.risk_warnings||[]).length,"预警")+tile("mint","PR 合并",tw.prs_merged||0,"本周")+tile("amber","trend",(o.trend&&o.trend.activity_level)||"—","活跃度");}
case "s6": {const tl=o.timeline||{}; const sm=a=>(a||[]).reduce((x,y)=>x+(y||0),0); return tile("indigo","weeks",(tl.labels||[]).length,"时间线周数")+tile("coral","commits",sm(tl.commits),"累计")+tile("mint","PRs",sm(tl.prs),"累计")+tile("amber","issues",sm(tl.issues),"累计");}
}
return defaultMetrics();
}
function setMetrics(html){document.getElementById("metrics").innerHTML=html;}
/* ---------- findings per scenario ---------- */
function find(cls,title,body,sub){return `<div class="find ${cls}"><div class="ft"><span>${title}</span></div><div class="fb">${body}</div>${sub?`<div class="fs">${sub}</div>`:""}</div>`;}
function rank(c){return `<div class="bar"><i style="width:${Math.min(100,Math.max(4,c))}%"></i></div>`;}
function findingsFor(key,o){
let out="";
switch(key){
case "s1": (o.innovation_points||[]).slice(0,6).forEach(i=>{out+=find("warning","["+i.category+"] 创新点",esc(i.description),esc(i.evidence));}); break;
case "s2": (o.topic_heat||[]).slice(0,4).forEach(h=>{out+=find("success","主题热度 · "+h.topic,"覆盖 "+h.count+" 个仓库",rank(h.count*15));});
(o.core_scholars||[]).slice(0,3).forEach(s=>{out+=find("",esc(s.login),"关联 "+s.repo_count+" 个仓库");}); break;
case "s3": (o.risk_items||o.risks||[]).slice(0,6).forEach(r=>{const lv=(r.level||"").includes("high")||r.severity==="P0"?"critical":"warning"; out+=find(lv,esc((r.level||r.severity||"")+" · "+(r.category||r.name||"")),esc(r.detail||r.evidence||r.name||""),esc(r.file?r.file:""));});
out+=find("success","复现性 "+(o.repro_score||0)+"/10 · 合规 "+(o.compliance_score||0)+"/10","许可证: "+esc(o.license?(typeof o.license==="string"?o.license:(o.license.id||"—")):"—")); break;
case "s4": (o.candidates||[]).slice(0,5).forEach((c,i)=>{out+=find(i===0?"success":"", "#"+(i+1)+" "+esc(c.login), (c.reasons||[]).join("")||"—", rank(c.score)+` <span style="float:right">${c.score}分</span>`);}); break;
case "s5": (o.risk_warnings||[]).slice(0,6).forEach(w=>{out+=find(w.level==="critical"?"critical":"warning",esc(w.level+" · "+w.type),esc(w.message),esc(w.suggestion||""));}); break;
case "s6": out+=find("success","可视化产物","时间线 "+((o.timeline&&o.timeline.labels||[]).length)+" 周 · 语言 "+((o.language_pie&&o.language_pie.length)||0)+" 类","已生成可交互 plotly 图表"); break;
}
return out||'<div class="empty">(该场景无结构化简报,详见完整结果页)</div>';
}
function setFindings(html){document.getElementById("findings").innerHTML=html;}
/* ---------- utilities ---------- */
function esc(s){return String(s).replace(/[&<>]/g,c=>({'&':'&amp;','<':'&lt;','>':'&gt;'}[c]));}
/* ---------- resize handler ---------- */
let resizeTimer;
window.addEventListener("resize",()=>{
clearTimeout(resizeTimer);
resizeTimer=setTimeout(()=>{
const src=ACTIVE&&ACTIVE._graphData?ACTIVE._graphData:DEMO;
drawGraph(src);
},200);
});
/* go */
init();
</script>
</body>
</html>

View File

@ -0,0 +1,504 @@
<!DOCTYPE html>
<html lang="zh-CN">
<head>
<meta charset="UTF-8">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>GitLink Research Atlas · 结果详情</title>
<link rel="preconnect" href="https://fonts.googleapis.com">
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
<link href="https://fonts.googleapis.com/css2?family=Cormorant+Garamond:wght@500;600;700&family=Inter:wght@400;500;600;700&family=JetBrains+Mono:wght@500;700&family=Noto+Serif+SC:wght@600;700&family=Noto+Sans+SC:wght@400;500;700&display=swap" rel="stylesheet">
<script src="https://cdn.jsdelivr.net/npm/mermaid@10/dist/mermaid.min.js"></script>
<style>
:root{
--bg:#F7F9FC; --card:#FFFFFF; --ink:#1E2A44; --ink2:#475569; --mute:#94A3B8;
--line:#E6EBF2; --indigo:#4F6BED; --coral:#F26B5E; --mint:#2EC4B6;
--t-blue:#E8EEFF; --t-coral:#FFEDEA; --t-mint:#E6F7F4; --t-amber:#FEF3C7;
--serif:"Cormorant Garamond","Noto Serif SC",Georgia,serif;
--sans:"Inter","Noto Sans SC",-apple-system,"PingFang SC","Microsoft YaHei",sans-serif;
--mono:"JetBrains Mono",ui-monospace,SFMono-Regular,Menlo,Consolas,monospace;
--shadow:0 1px 2px rgba(30,42,68,.04),0 6px 18px rgba(30,42,68,.06);
--shadow-sm:0 1px 2px rgba(30,42,68,.06);
}
*{box-sizing:border-box;margin:0;padding:0}
html,body{height:auto;min-height:100%;margin:0}
body{font-family:var(--sans);color:var(--ink);background:var(--bg);
background-image:radial-gradient(1200px 480px at 80% -8%,#EAF0FF 0%,rgba(234,240,255,0) 60%),
radial-gradient(900px 420px at 0% 0%,#FFF1EE 0%,rgba(255,241,238,0) 55%);
-webkit-font-smoothing:antialiased}
/* ========== PRINT ========== */
@media print{
body{background:#fff!important}
.no-print{display:none!important}
.page-wrap{max-width:100%!important;box-shadow:none!important}
}
/* ========== PAGE WRAPPER ========== */
.page-wrap{max-width:1120px;margin:0 auto;padding:32px 40px 60px}
@media(max-width:760px){.page-wrap{padding:20px 16px 40px}}
/* ========== HEADER ========== */
.res-header{margin-bottom:32px}
.res-header .back{display:inline-flex;align-items:center;gap:6px;font-size:13px;font-weight:600;
color:var(--indigo);text-decoration:none;margin-bottom:16px;padding:6px 14px;border-radius:10px;
background:var(--t-blue);transition:.15s}
.res-header .back:hover{background:#D6DEFF}
.res-header h1{font-family:var(--serif);font-weight:700;font-size:34px;line-height:1.2;letter-spacing:.3px;color:var(--ink)}
.res-header .sub{font-size:15px;color:var(--ink2);margin-top:6px;line-height:1.5}
.res-header .meta{display:flex;flex-wrap:wrap;gap:16px;margin-top:14px;align-items:center}
.res-header .meta .chip{display:inline-flex;align-items:center;gap:6px;font-size:12px;font-weight:500;
padding:5px 12px;border-radius:999px;background:var(--card);border:1px solid var(--line);color:var(--ink2);box-shadow:var(--shadow-sm)}
.res-header .meta .chip b{font-family:var(--mono);font-weight:700;color:var(--ink)}
.res-header .cmd-drawer{margin-top:14px;background:var(--ink);color:#CBD5E1;font-family:var(--mono);
font-size:12px;padding:10px 16px;border-radius:10px;display:flex;align-items:center;gap:10px;overflow:hidden}
.res-header .cmd-drawer .dots{display:flex;gap:6px;flex-shrink:0}
.res-header .cmd-drawer .dots i{width:11px;height:11px;border-radius:50%;display:block}
.res-header .cmd-drawer .cmd{flex:1;white-space:nowrap;overflow:hidden;text-overflow:ellipsis;color:#E2E8F0}
/* ========== SECTION TITLES ========== */
.section{margin-top:36px}
.section-title{font-family:var(--serif);font-weight:700;font-size:22px;color:var(--ink);margin-bottom:16px;
padding-bottom:8px;border-bottom:2px solid var(--line)}
.section-title .eyebrow{font-family:var(--mono);font-size:10px;letter-spacing:.16em;text-transform:uppercase;
color:var(--mute);display:block;margin-bottom:4px}
/* ========== METRICS ROW ========== */
.metrics-row{display:grid;grid-template-columns:repeat(4,1fr);gap:14px;margin-bottom:8px}
@media(max-width:760px){.metrics-row{grid-template-columns:repeat(2,1fr)}}
.metric-card{background:var(--card);border:1px solid var(--line);border-radius:12px;padding:18px;
box-shadow:var(--shadow);text-align:center}
.metric-card .label{font-size:12px;color:var(--ink2);font-weight:500;margin-bottom:8px}
.metric-card .value{font-family:var(--mono);font-weight:700;font-size:32px;color:var(--ink);line-height:1}
.metric-card .note{font-size:11px;color:var(--mute);margin-top:6px}
/* ========== KNOWLEDGE GRAPH ========== */
.graph-container{background:var(--card);border:1px solid var(--line);border-radius:14px;box-shadow:var(--shadow);
padding:20px;overflow:hidden}
.graph-svg-wrap{width:100%;height:500px;position:relative;background:radial-gradient(500px 380px at 50% 45%,#F2F6FF 0%,rgba(255,255,255,0) 70%);
border-radius:10px;overflow:hidden}
.graph-svg-wrap svg{width:100%;height:100%;display:block}
.g-edge{stroke:#9DB4F0;stroke-opacity:.55;stroke-width:1}
.g-node circle{stroke:#fff;stroke-width:2;transition:.2s}
.g-node:hover circle{filter:brightness(1.08)}
.g-node text{font-family:var(--sans);font-size:11px;font-weight:600;fill:var(--ink);pointer-events:none;text-anchor:middle}
.graph-legend{display:flex;flex-wrap:wrap;gap:14px;margin-top:14px;padding-top:12px;border-top:1px solid var(--line)}
.graph-legend .leg-item{display:flex;align-items:center;gap:5px;font-size:12px;color:var(--ink2)}
.graph-legend .leg-dot{width:10px;height:10px;border-radius:50%;flex-shrink:0}
.graph-legend .leg-note{font-size:12px;color:var(--mute);margin-left:auto}
/* ========== VIZ IFRAME (S6) ========== */
.viz-container{background:var(--card);border:1px solid var(--line);border-radius:14px;box-shadow:var(--shadow);
padding:12px;overflow:hidden}
.viz-container iframe{width:100%;height:520px;border:0;border-radius:10px;background:#fff}
/* ========== MERMAID ========== */
.mmd-container{background:var(--card);border:1px solid var(--line);border-radius:14px;box-shadow:var(--shadow);
padding:16px;overflow:auto;max-height:600px}
/* ========== RADAR / SCORE (S3) ========== */
.score-display{display:flex;align-items:center;justify-content:center;gap:40px;padding:30px;
background:var(--card);border:1px solid var(--line);border-radius:14px;box-shadow:var(--shadow);flex-wrap:wrap}
.score-circle{width:140px;height:140px;border-radius:50%;display:flex;flex-direction:column;align-items:center;justify-content:center;
position:relative;border:6px solid}
.score-circle .val{font-family:var(--mono);font-weight:700;font-size:40px;line-height:1}
.score-circle .label{font-size:13px;color:var(--ink2);margin-top:2px}
.score-circle.repro{border-color:var(--mint);color:var(--mint);background:var(--t-mint)}
.score-circle.compliance{border-color:var(--indigo);color:var(--indigo);background:var(--t-blue)}
/* ========== WEEKLY COMPARISON (S5) ========== */
.weekly-bars{display:flex;flex-direction:column;gap:16px;padding:20px;
background:var(--card);border:1px solid var(--line);border-radius:14px;box-shadow:var(--shadow)}
.week-row{display:flex;align-items:center;gap:14px}
.week-row .label{width:70px;font-size:13px;font-weight:600;color:var(--ink2);flex-shrink:0}
.week-row .bar-bg{flex:1;height:28px;background:#F1F5F9;border-radius:8px;overflow:hidden;position:relative}
.week-row .bar-fill{height:100%;border-radius:8px;transition:width .6s ease}
.week-row .bar-fill.this-week{background:linear-gradient(90deg,var(--indigo),#7C8CF5)}
.week-row .bar-fill.last-week{background:var(--line)}
.week-row .bar-val{position:absolute;right:8px;top:50%;transform:translateY(-50%);font-family:var(--mono);font-size:12px;font-weight:600;color:var(--ink)}
.risk-list{margin-top:20px;display:flex;flex-direction:column;gap:10px}
.risk-item{padding:10px 14px;border-radius:10px;border-left:4px solid}
.risk-item.critical{border-color:var(--coral);background:var(--t-coral)}
.risk-item.warning{border-color:#F59E0B;background:var(--t-amber)}
.risk-item .type{font-size:11px;font-family:var(--mono);color:var(--mute);text-transform:uppercase}
.risk-item .msg{font-size:13px;color:var(--ink);margin-top:4px}
.risk-item .suggestion{font-size:12px;color:var(--ink2);margin-top:4px;font-style:italic}
/* ========== REPORT (markdown) ========== */
.report-content{background:var(--card);border:1px solid var(--line);border-radius:14px;box-shadow:var(--shadow);
padding:28px 32px;line-height:1.75;font-size:14px;color:var(--ink)}
.report-content h1{font-family:var(--serif);font-weight:700;font-size:26px;margin:24px 0 12px;padding-bottom:8px;border-bottom:2px solid var(--line)}
.report-content h1:first-child{margin-top:0}
.report-content h2{font-family:var(--serif);font-weight:700;font-size:20px;margin:20px 0 10px;color:var(--ink)}
.report-content h3{font-weight:700;font-size:16px;margin:16px 0 8px;color:var(--ink2)}
.report-content p{margin:10px 0}
.report-content ul,.report-content ol{margin:8px 0;padding-left:24px}
.report-content li{margin:4px 0}
.report-content strong{font-weight:700;color:var(--ink)}
.report-content code{font-family:var(--mono);font-size:12.5px;background:var(--t-blue);padding:2px 6px;border-radius:4px;color:var(--indigo)}
.report-content table{width:100%;border-collapse:collapse;margin:12px 0;font-size:13px}
.report-content th{background:var(--t-blue);font-weight:600;text-align:left;padding:8px 12px;border:1px solid var(--line)}
.report-content td{padding:8px 12px;border:1px solid var(--line)}
/* ========== STRUCTURED CARDS ========== */
.cards-grid{display:grid;grid-template-columns:repeat(auto-fill,minmax(320px,1fr));gap:16px;margin-top:20px}
.conclusion-card{background:var(--card);border:1px solid var(--line);border-radius:12px;box-shadow:var(--shadow);
padding:16px 18px;border-top:4px solid}
.conclusion-card.innovation{border-top-color:var(--indigo)}
.conclusion-card.risk{border-top-color:var(--coral)}
.conclusion-card.collab{border-top-color:var(--mint)}
.conclusion-card .card-title{font-weight:700;font-size:14px;color:var(--ink);margin-bottom:8px}
.conclusion-card .card-body{font-size:13px;color:var(--ink2);line-height:1.55}
.conclusion-card .card-body p{margin:4px 0}
/* ========== LOADING STATE ========== */
.loading-state{display:flex;flex-direction:column;align-items:center;justify-content:center;padding:80px 20px;color:var(--mute);gap:16px}
.spin-lg{width:40px;height:40px;border-radius:50%;border:4px solid var(--t-blue);border-top-color:var(--indigo);animation:rot .8s linear infinite}
@keyframes rot{to{transform:rotate(360deg)}}
/* ========== ERROR ========== */
.error-state{padding:60px 20px;text-align:center;color:var(--coral)}
.error-state h2{font-family:var(--serif);font-size:24px;margin-bottom:12px}
.error-state p{font-size:14px;color:var(--ink2);max-width:500px;margin:0 auto}
/* empty */
.empty-state{padding:60px 20px;text-align:center;color:var(--mute)}
.empty-state p{font-size:14px}
</style>
</head>
<body>
<div class="page-wrap" id="app">
<div class="loading-state"><div class="spin-lg"></div><div>正在加载结果…</div></div>
</div>
<script>
mermaid.initialize({startOnLoad:false, theme:"base", securityLevel:"loose",
themeVariables:{primaryColor:"#E8EEFF",primaryTextColor:"#1E2A44",primaryBorderColor:"#4F6BED",lineColor:"#9DB4F0",fontSize:"13px"}});
const NODE_COLOR={repo:"#4F6BED",scholar:"#3B82F6",team:"#6366F1",topic:"#06B6D4",method:"#6366F1",
paper:"#8B5CF6",dataset:"#2EC4B6",model:"#4F6BED",experiment:"#F26B5E",
reproduce:"#2EC4B6",license:"#64748B",trend:"#F26B5E",issue:"#F59E0B",pr:"#10B981",default:"#94A3B8"};
const NODE_LABEL={repo:"仓库",scholar:"学者",topic:"主题",paper:"论文",dataset:"数据集",
model:"模型",experiment:"实验",reproduce:"复现",license:"许可证",trend:"趋势",issue:"Issue",pr:"PR"};
function esc(s){return String(s).replace(/[&<>]/g,c=>({'&':'&amp;','<':'&lt;','>':'&gt;'}[c]));}
function token(){return localStorage.getItem("demo_token")||"";}
function licOf(o){const l=o.license;if(!l)return o.license_id||"—";return typeof l==="string"?l:(l.id||l.name||l.spdx_id||l.identifier||"—");}
/* ========== Force layout (same as index) ========== */
function forceLayout(nodes,edges,W,H,iters){
const cx=W/2,cy=H/2;
nodes.forEach(n=>{if(n.x==null){n.x=cx+(Math.random()-.5)*W*.5;n.y=cy+(Math.random()-.5)*H*.5;}});
for(let it=0;it<iters;it++){
for(const n of nodes){n._fx=0;n._fy=0;}
for(let i=0;i<nodes.length;i++){for(let j=i+1;j<nodes.length;j++){
let dx=nodes[i].x-nodes[j].x,dy=nodes[i].y-nodes[j].y,d2=dx*dx+dy*dy+.02,d=Math.sqrt(d2),f=2600/d2,fx=dx/d*f,fy=dy/d*f;
nodes[i]._fx+=fx;nodes[i]._fy+=fy;nodes[j]._fx-=fx;nodes[j]._fy-=fy;}}
for(const e of edges){const a=nodes.find(n=>n.id===e[0]),b=nodes.find(n=>n.id===e[1]);if(!a||!b)continue;
let dx=b.x-a.x,dy=b.y-a.y,d=Math.sqrt(dx*dx+dy*dy)+.02,L=a.imp||b.imp?130:85,k=.04;
let f=(d-L)*k,fx=dx/d*f,fy=dy/d*f;a._fx+=fx;a._fy+=fy;b._fx-=fx;b._fy-=fy;}
for(const n of nodes){n._fx+=(cx-n.x)*.012;n._fy+=(cy-n.y)*.012;
n.x+=Math.max(-14,Math.min(14,n._fx));n.y+=Math.max(-14,Math.min(14,n._fy));
n.x=Math.max(45,Math.min(W-45,n.x));n.y=Math.max(38,Math.min(H-34,n.y));}}
}
function drawGraphIn(containerId,data){
const svg=document.getElementById(containerId); if(!svg) return;
const W=svg.clientWidth||800,H=svg.clientHeight||500;
svg.setAttribute("viewBox",`0 0 ${W} ${H}`);
let nodes=(data.nodes||[]).map(n=>Object.assign({},n));
let edges=(data.edges||[]).map(e=>Array.isArray(e)?e:[e.source||e.from,e.target||e.to]);
if(nodes.length>40){
const deg=new Map(nodes.map(n=>[n.id,0]));
edges.forEach(e=>{deg.set(e[0],(deg.get(e[0])||0)+1);deg.set(e[1],(deg.get(e[1])||0)+1);});
nodes.sort((a,b)=>(deg.get(b.id)||0)-(deg.get(a.id)||0));
const keep=new Set(nodes.slice(0,40).map(n=>n.id));nodes=nodes.filter(n=>keep.has(n.id));edges=edges.filter(e=>keep.has(e[0])&&keep.has(e[1]));
}
forceLayout(nodes,edges,W,H,300);
let s=`<defs><filter id="glow2"><feGaussianBlur stdDeviation="5" result="blur"/><feMerge><feMergeNode in="blur"/><feMergeNode in="SourceGraphic"/></feMerge></filter></defs>`;
for(const e of edges){const a=nodes.find(n=>n.id===e[0]),b=nodes.find(n=>n.id===e[1]);if(!a||!b)continue;
s+=`<line class="g-edge" x1="${a.x.toFixed(1)}" y1="${a.y.toFixed(1)}" x2="${b.x.toFixed(1)}" y2="${b.y.toFixed(1)}"/>`;}
for(const n of nodes){const col=NODE_COLOR[n.type]||NODE_COLOR.default,r=n.imp?13:8;
s+=`<g class="g-node" transform="translate(${n.x.toFixed(1)},${n.y.toFixed(1)})"><circle r="${r}" fill="${col}"${n.imp?' filter="url(#glow2)"':""}/><text y="${r+13}">${esc(n.label||n.id).slice(0,18)}</text></g>`;}
svg.innerHTML=s;
}
function buildGraphLegend(types){
let h="";
for(const t of [...types].sort()){
h+=`<span class="leg-item"><span class="leg-dot" style="background:${NODE_COLOR[t]||NODE_COLOR.default}"></span>${NODE_LABEL[t]||t}</span>`;
}
return h;
}
/* ========== Simple markdown to HTML ========== */
function mdToHtml(md){
if(!md) return "";
let html=esc(md);
// code blocks
html=html.replace(/```(\w*)\n([\s\S]*?)```/g,'<pre><code>$2</code></pre>');
// inline code
html=html.replace(/`([^`]+)`/g,'<code>$1</code>');
// tables
html=html.replace(/^(\|.+\|)\n(\|[-:\s|]+\|)\n((?:\|.+\|\n?)*)/gm,function(_,hdr,sep,body){
const ths=hdr.split("|").filter(c=>c.trim()).map(c=>`<th>${c.trim()}</th>`).join("");
let rows="";
body.trim().split("\n").forEach(row=>{
const tds=row.split("|").filter(c=>c.trim()).map(c=>`<td>${c.trim()}</td>`).join("");
rows+=`<tr>${tds}</tr>`;
});
return `<table><thead><tr>${ths}</tr></thead><tbody>${rows}</tbody></table>`;
});
// headings
html=html.replace(/^### (.+)$/gm,'<h3>$1</h3>');
html=html.replace(/^## (.+)$/gm,'<h2>$1</h2>');
html=html.replace(/^# (.+)$/gm,'<h1>$1</h1>');
// bold
html=html.replace(/\*\*([^*]+)\*\*/g,'<strong>$1</strong>');
// unordered lists
html=html.replace(/^- (.+)$/gm,'<li>$1</li>');
html=html.replace(/(<li>.*<\/li>\n?)+/g,'<ul>$&</ul>');
// paragraphs (lines not inside tags)
html=html.replace(/^(?!<[huplo])((?!<).+)$/gm,'<p>$1</p>');
return html;
}
/* ========== MAIN RENDER ========== */
async function loadResult(){
const params=new URLSearchParams(window.location.search);
const scenario=params.get("scenario");
if(!scenario){
document.getElementById("app").innerHTML='<div class="error-state"><h2>缺少场景参数</h2><p>请从仪表盘点击场景,或手动打开 <code>result.html?scenario=s1</code></p></div>';
return;
}
let data;
try{
const r=await fetch("/api/result/"+encodeURIComponent(scenario),{headers:{"X-Demo-Token":token()}});
data=await r.json();
}catch(e){
document.getElementById("app").innerHTML='<div class="error-state"><h2>加载失败</h2><p>'+esc(e.message)+'</p><a class="back" href="index.html">返回仪表盘</a></div>';
return;
}
if(!data.ok){
document.getElementById("app").innerHTML='<div class="error-state"><h2>'+esc(data.error||"数据未就绪")+'</h2><p>请先在仪表盘运行该场景以生成结果。</p><a class="back" href="index.html">返回仪表盘</a></div>';
return;
}
const art=data.artifacts||{};
const obj=art.json?JSON.parse(art.json):null;
const owner=params.get("owner")||val_from("owner")||"";
const repo=params.get("repo")||val_from("repo")||"";
const keyword=params.get("keyword")||val_from("keyword")||"";
render(data,obj,art,scenario,owner,repo,keyword);
}
function val_from(name){try{return localStorage.getItem("atlas_"+name)||"";}catch(e){return"";}}
function render(data,obj,art,scenario,owner,repo,keyword){
const label=data.label||scenario;
const desc=data.desc||"";
const now=new Date().toLocaleString("zh-CN");
let h="";
// ---- HEADER ----
h+=`<div class="res-header">
<a class="back no-print" href="index.html">&#8592; 返回仪表盘</a>
<h1>${esc(label)}</h1>
<div class="sub">${esc(desc)}</div>
<div class="meta">`;
if(owner||repo) h+=`<span class="chip">仓库 <b>${esc(owner)}${repo?"/"+esc(repo):""}</b></span>`;
if(keyword) h+=`<span class="chip">关键词 <b>${esc(keyword)}</b></span>`;
h+=`<span class="chip">场景 <b>${esc(scenario)}</b></span>
<span class="chip">生成于 ${esc(now)}</span>
</div>
<div class="cmd-drawer no-print">
<div class="dots"><i style="background:#FF5F57"></i><i style="background:#FEBC2E"></i><i style="background:#28C840"></i></div>
<span style="color:var(--mint)">atlas &#10095;</span>
<span class="cmd">gitlink-cli research --scenario ${esc(scenario)} ${owner?"--repo "+esc(owner)+"/"+esc(repo):""} ${keyword?'-k "'+esc(keyword)+'"':""}</span>
</div></div>`;
// ---- METRICS ROW ----
const metrics=metricsFor(scenario,obj);
if(metrics){
h+=`<div class="section"><div class="section-title"><span class="eyebrow">Key Metrics</span>关键指标</div>
<div class="metrics-row">${metrics}</div></div>`;
}
// ---- MAIN VIZ ----
h+=`<div class="section"><div class="section-title"><span class="eyebrow">Visualization</span>可视化</div>`;
// S2: Knowledge Graph
if(scenario==="s2"&&obj&&obj.nodes&&obj.edges){
const graphNodes=obj.nodes.map(n=>({id:n.id,type:n.type||"default",label:n.label||n.id,imp:false}));
// highlight repo and high-degree nodes
const deg=new Map();
obj.edges.forEach(e=>{deg.set((e.source||e.from),(deg.get(e.source||e.from)||0)+1);deg.set((e.target||e.to),(deg.get(e.target||e.to)||0)+1);});
graphNodes.forEach(n=>{if(n.type==="repo"||deg.get(n.id)>3)n.imp=true;});
const graphEdges=obj.edges.map(e=>[e.source||e.from,e.target||e.to]);
const types=new Set(graphNodes.map(n=>n.type));
const gId="resGraph_"+Date.now();
h+=`<div class="graph-container">
<div class="graph-svg-wrap"><svg id="${gId}" preserveAspectRatio="xMidYMid meet"></svg></div>
<div class="graph-legend">${buildGraphLegend(types)}
<span class="leg-note">${esc(label)} — ${(obj.meta||{}).repo_count||graphNodes.length} 个仓库节点, ${(obj.meta||{}).edge_count||graphEdges.length} 条关联边</span>
</div></div>`;
// defer drawing
setTimeout(()=>drawGraphIn(gId,{nodes:graphNodes,edges:graphEdges}),100);
}
// S6: plotly
else if(scenario==="s6"&&art.html){
const blob=URL.createObjectURL(new Blob([art.html],{type:"text/html"}));
h+=`<div class="viz-container"><iframe src="${blob}"></iframe></div>`;
}
// S1/S4: mermaid
else if((scenario==="s1"||scenario==="s4")&&art.mermaid){
const mId="resMmd_"+Date.now();
const mmd=art.mermaid.replace(/```mermaid|```/g,"");
h+=`<div class="mmd-container"><div class="mermaid" id="${mId}">${esc(mmd)}</div></div>`;
setTimeout(()=>{const el=document.getElementById(mId);if(el)el.removeAttribute("data-processed");
mermaid.run({nodes:["#"+mId]}).catch(()=>{});},100);
}
// S3: radar / scores
else if(scenario==="s3"&&obj){
const rs=obj.repro_score||0;const cs=obj.compliance_score||0;
h+=`<div class="score-display">
<div class="score-circle repro"><span class="val">${rs}</span><span class="label">复现性 / 10</span></div>
<div class="score-circle compliance"><span class="val">${cs}</span><span class="label">合规性 / 10</span></div>
</div>`;
}
// S5: weekly comparison + risks
else if(scenario==="s5"&&obj){
const ws=obj.week_stats||{};
const tw=ws.this_week||{};const lw=ws.last_week||{};
const maxC=Math.max(tw.commits||0,lw.commits||0,1);
const maxP=Math.max(tw.prs_merged||0,lw.prs_merged||0,1);
const maxI=Math.max(tw.issues_closed||0,lw.issues_closed||0,1);
h+=`<div class="weekly-bars">
<div style="font-weight:700;font-size:14px;margin-bottom:8px;color:var(--ink)">本周 vs 上周 对比</div>
<div class="week-row"><span class="label">Commits</span><div class="bar-bg"><div class="bar-fill last-week" style="width:${(lw.commits/maxC*100||0).toFixed(1)}%"><span class="bar-val">${lw.commits||0}</span></div><div class="bar-fill this-week" style="width:${(tw.commits/maxC*100||0).toFixed(1)}%;position:relative;margin-top:-28px"><span class="bar-val">${tw.commits||0}</span></div></div></div>
<div class="week-row"><span class="label">PR merged</span><div class="bar-bg"><div class="bar-fill last-week" style="width:${(lw.prs_merged/maxP*100||0).toFixed(1)}%"><span class="bar-val">${lw.prs_merged||0}</span></div><div class="bar-fill this-week" style="width:${(tw.prs_merged/maxP*100||0).toFixed(1)}%;position:relative;margin-top:-28px"><span class="bar-val">${tw.prs_merged||0}</span></div></div></div>
<div class="week-row"><span class="label">Issues</span><div class="bar-bg"><div class="bar-fill last-week" style="width:${(lw.issues_closed/maxI*100||0).toFixed(1)}%"><span class="bar-val">${lw.issues_closed||0}</span></div><div class="bar-fill this-week" style="width:${(tw.issues_closed/maxI*100||0).toFixed(1)}%;position:relative;margin-top:-28px"><span class="bar-val">${tw.issues_closed||0}</span></div></div></div>
</div>`;
// risk warnings
const risks=obj.risk_warnings||[];
if(risks.length){
h+=`<div style="margin-top:20px;font-weight:700;font-size:14px;color:var(--ink)">风险预警</div><div class="risk-list">`;
risks.forEach(r=>{const cls=r.level==="critical"?"critical":"warning";
h+=`<div class="risk-item ${cls}"><div class="type">${esc(r.level||"")} · ${esc(r.type||"")}</div><div class="msg">${esc(r.message||"")}</div>${r.suggestion?`<div class="suggestion">${esc(r.suggestion)}</div>`:""}</div>`;});
h+=`</div>`;
}
}
else{
h+=`<div class="empty-state"><p>该场景无可视化产物,详见下方报告。</p></div>`;
}
h+=`</div>`;
// ---- REPORT ----
if(art.report){
h+=`<div class="section"><div class="section-title"><span class="eyebrow">Full Report</span>完整报告</div>
<div class="report-content">${mdToHtml(art.report)}</div></div>`;
}
// ---- STRUCTURED CONCLUSION CARDS ----
const conclusions=buildConclusionCards(scenario,obj);
if(conclusions){
h+=`<div class="section"><div class="section-title"><span class="eyebrow">Conclusions</span>结构化结论</div>
<div class="cards-grid">${conclusions}</div></div>`;
}
document.getElementById("app").innerHTML=h;
}
/* ========== METRICS ========== */
function metricCard(label,val,note){
return `<div class="metric-card"><div class="label">${esc(label)}</div><div class="value">${esc(String(val))}</div><div class="note">${esc(note||"")}</div></div>`;
}
function metricsFor(key,o){
if(!o) return "";
const M=o.meta||{};
switch(key){
case "s1": return metricCard("commits",M.commit_count||0,"采样提交")+metricCard("merged PR",M.merged_pr_count||0,"已合并")+metricCard("docs",M.doc_count||0,"文档文件")+metricCard("创新点",(o.innovation_points||[]).length,"高影响合并");
case "s2": return metricCard("repos",M.repo_count||(o.nodes||[]).length,"仓库节点")+metricCard("scholars",M.scholar_count||0,"学者")+metricCard("topics",M.topic_count||0,"主题方向")+metricCard("edges",M.edge_count||(o.edges||[]).length,"关系边");
case "s3": return metricCard("复现性",(o.repro_score||0)+"/10","reproducibility")+metricCard("合规性",(o.compliance_score||0)+"/10","compliance")+metricCard("风险项",(o.risks||o.risk_items||[]).length,"项")+metricCard("许可证",esc(licOf(o)),"识别");
case "s4": return metricCard("缺口主题",(o.gap_topics||[]).length,"gap topics")+metricCard("推荐候选",(o.candidates||[]).length,"candidates")+metricCard("最高匹配",o.candidates&&o.candidates[0]?o.candidates[0].score:0,"top score")+metricCard("候选池",o.meta&&o.meta.pool_size||0,"pool size");
case "s5":{const tw=(o.week_stats&&o.week_stats.this_week)||{};return metricCard("本周 commits",tw.commits||0,"")+metricCard("预警",(o.risk_warnings||[]).length,"risks")+metricCard("PR 合并",tw.prs_merged||0,"")+metricCard("活跃度",(o.trend&&o.trend.activity_level)||"—","");}
case "s6":{const tl=o.timeline||{};const sm=a=>(a||[]).reduce((x,y)=>x+(y||0),0);return metricCard("周数",(tl.labels||[]).length,"")+metricCard("commits",sm(tl.commits),"")+metricCard("PRs",sm(tl.prs),"")+metricCard("issues",sm(tl.issues),"");}
}
return "";
}
/* ========== CONCLUSION CARDS ========== */
function buildConclusionCards(key,o){
if(!o) return "";
let cards="";
switch(key){
case "s1":{
const ips=o.innovation_points||[];
if(ips.length){
cards+=`<div class="conclusion-card innovation"><div class="card-title">&#128161; 创新点 (${ips.length})</div><div class="card-body">`;
ips.slice(0,8).forEach(i=>{cards+=`<p><strong>[${esc(i.category)}]</strong> ${esc(i.description)}<br><small style="color:var(--mute)">${esc(i.evidence||"")}</small></p>`;});
cards+=`</div></div>`;
}
break;}
case "s2":{
const scholars=o.core_scholars||[];
if(scholars.length){
cards+=`<div class="conclusion-card collab"><div class="card-title">&#128101; 核心学者</div><div class="card-body">`;
scholars.slice(0,6).forEach(s=>{cards+=`<p><strong>${esc(s.login)}</strong> — ${s.repo_count} 个仓库</p>`;});
cards+=`</div></div>`;
}
const topics=o.topic_heat||[];
if(topics.length){
cards+=`<div class="conclusion-card innovation"><div class="card-title">&#128293; 热门主题</div><div class="card-body">`;
topics.slice(0,6).forEach(t=>{cards+=`<p><strong>${esc(t.topic)}</strong> — ${t.count} 个仓库</p>`;});
cards+=`</div></div>`;
}
break;}
case "s3":{
const risks=o.risk_items||o.risks||[];
if(risks.length){
cards+=`<div class="conclusion-card risk"><div class="card-title">&#9888;&#65039; 风险项 (${risks.length})</div><div class="card-body">`;
risks.slice(0,8).forEach(r=>{
const lvl=(r.level||"").includes("high")||r.severity==="P0"?"&#x1F534;":"&#x1F7E1;";
cards+=`<p>${lvl} <strong>${esc(r.category||r.name||"")}</strong> — ${esc(r.detail||r.evidence||"")}</p>`;});
cards+=`</div></div>`;
}
break;}
case "s4":{
const cands=o.candidates||[];
if(cands.length){
cards+=`<div class="conclusion-card collab"><div class="card-title">&#129309; 协作推荐 (${cands.length})</div><div class="card-body">`;
cands.slice(0,6).forEach((c,i)=>{cards+=`<p><strong>#${i+1} ${esc(c.login)}</strong> (${c.score}分)<br><small>${(c.reasons||[]).join("")}</small></p>`;});
cards+=`</div></div>`;
}
const gaps=o.gap_topics||[];
if(gaps.length){
cards+=`<div class="conclusion-card innovation"><div class="card-title">&#128270; 缺口主题 (${gaps.length})</div><div class="card-body">`;
gaps.slice(0,6).forEach(g=>{cards+=`<p>${esc(typeof g==="string"?g:(g.topic||g.name||JSON.stringify(g)))}</p>`;});
cards+=`</div></div>`;
}
break;}
case "s5":{
const risks=o.risk_warnings||[];
if(risks.length){
cards+=`<div class="conclusion-card risk"><div class="card-title">&#9888;&#65039; 风险预警 (${risks.length})</div><div class="card-body">`;
risks.slice(0,6).forEach(w=>{cards+=`<p><strong>[${esc(w.type)}]</strong> ${esc(w.message)}${w.suggestion?"<br><small>"+esc(w.suggestion)+"</small>":""}</p>`;});
cards+=`</div></div>`;
}
const trend=o.trend||{};
cards+=`<div class="conclusion-card innovation"><div class="card-title">&#128200; 活跃趋势</div><div class="card-body"><p>commit 变化: <strong>${trend.commit_delta_pct||"—"}</strong></p><p>活跃度: <strong>${esc(trend.activity_level||"—")}</strong></p></div></div>`;
break;}
case "s6":{
const tl=o.timeline||{};
cards+=`<div class="conclusion-card innovation"><div class="card-title">&#128202; 数据概览</div><div class="card-body"><p>时间线: ${(tl.labels||[]).length} 周</p><p>语言分布: ${(o.language_pie||[]).length} 类</p><p>累计 commits: ${(tl.commits||[]).reduce((a,b)=>a+(b||0),0)}</p></div></div>`;
break;}
}
return cards;
}
/* go */
loadResult();
</script>
</body>
</html>

View File

@ -0,0 +1,68 @@
# Issue batch maintenance shortcuts
## Summary
Add OpenAPI-backed Issue batch maintenance shortcuts:
- `issue +batch-update` — batch update Issue status, priority, milestone, tags, and assigners by API issue IDs.
- `issue +batch-delete` — batch delete Issues by API issue IDs with explicit confirmation.
This complements the existing `issue +batch-close` command. `batch-close` uses web URL issue numbers, while the OpenAPI batch update/delete endpoints use API issue IDs.
## OpenAPI coverage
| Command | Method | Endpoint |
|---|---|---|
| `issue +batch-update` | PATCH | `/api/v1/{owner}/{repo}/issues/batch_update.json` |
| `issue +batch-delete` | DELETE | `/api/v1/{owner}/{repo}/issues/batch_destroy.json` |
## ID semantics
- `issue +batch-close --numbers` uses web URL Issue numbers (`project_issues_index`).
- `issue +batch-update --ids` and `issue +batch-delete --ids` use API Issue IDs returned by Issue APIs.
The docs and help text explicitly call this out to avoid mixing the two ID types.
## Safety and usability
- Both commands support `--dry-run`.
- `issue +batch-update` requires at least one update field.
- `issue +batch-delete` is destructive and requires `--yes` for real execution.
- ID lists are validated as positive integers and de-duplicated.
## Examples
```bash
gitlink-cli issue +batch-update \
--owner Gitlink \
--repo forgeplus \
--ids 101,102 \
--status-id 3 \
--priority-id 2 \
--tag-ids 7,8 \
--assigner-ids 11,12 \
--dry-run
gitlink-cli issue +batch-delete \
--owner Gitlink \
--repo forgeplus \
--ids 101,102 \
--dry-run
gitlink-cli issue +batch-delete \
--owner Gitlink \
--repo forgeplus \
--ids 101,102 \
--yes
```
## Tests
```bash
GOPROXY=https://goproxy.cn,direct go test ./...
go vet ./...
go run . issue +batch-update --help
go run . issue +batch-delete --help
go run . issue +batch-update --owner wangyue111 --repo gitlink-cli --ids 101,102 --status-id 3 --dry-run --format json
go run . issue +batch-delete --owner wangyue111 --repo gitlink-cli --ids 101,102 --dry-run --format json
```

View File

@ -0,0 +1,23 @@
# Wiki Shortcut
新增 `wiki` Shortcut 组,支持 Wiki 页面管理:
- `wiki +list` - 列出 Wiki 页面(目录结构)
- `wiki +view` - 按页面名称查看 Wiki 页面详情
- `wiki +create` - 创建新的 Wiki 页面
- `wiki +update` - 更新 Wiki 页面标题和/或内容
- `wiki +delete` - 删除 Wiki 页面
## 实现要点
- **API 端点**:基于 `/api/wiki/open/{action}` 扁平路径结构,覆盖 5 个 Wiki 管理接口:
- `GET /api/wiki/open/wikiPages` — 目录列表
- `GET /api/wiki/open/getWiki` — 查看页面
- `POST /api/wiki/open/createWiki` — 创建页面
- `PUT /api/wiki/open/updateWiki` — 更新页面
- `DELETE /api/wiki/open/deleteWiki` — 删除页面
- **标识方式**Wiki 页面通过 `pageName`slug标识所有操作需要 `projectId`GitLink 项目数字 ID
- **内容编码**:创建和更新时,内容自动进行 base64 编码后以 `content_base64` 字段发送
- **更新保护**`+update` 要求必须提供 `--title``--page-name``--content` 为可选
- **Shortcut 模式**:使用 `common.Shortcut` + `RuntimeContext` 框架,与其他模块保持一致

View File

@ -43,6 +43,7 @@ func New() (*Client, error) {
// Do makes an API call with automatic .json suffix appended.
func (c *Client) Do(method, path string, body interface{}, query url.Values) (*output.Envelope, error) {
path = normalizeAPIPath(c.BaseURL, path)
return c.do(method, path, body, query, true, "json")
}
@ -62,10 +63,10 @@ func (c *Client) do(method, path string, body interface{}, query url.Values, app
if idx := strings.Index(path, "?"); idx != -1 {
basePath := path[:idx]
queryStr := path[idx:]
if !strings.HasSuffix(basePath, ".json") {
if shouldAppendJSONSuffix(basePath) {
path = basePath + ".json" + queryStr
}
} else if !strings.HasSuffix(path, ".json") {
} else if shouldAppendJSONSuffix(path) {
path += ".json"
}
}
@ -144,22 +145,36 @@ func (c *Client) do(method, path string, body interface{}, query url.Values, app
return output.SuccessEnvelope(string(respData), nil), nil
}
// Check GitLink error-in-body pattern
// Support both {"status":N, "message":"..."} and gateway {"code":N, "msg":"..."}
var bodyCode float64
var bodyMsg string
if status, ok := raw["status"]; ok {
var statusCode float64
switch v := status.(type) {
case float64:
statusCode = v
bodyCode = v
case int:
statusCode = float64(v)
bodyCode = float64(v)
}
if statusCode != 0 && statusCode != 200 && statusCode != 1 {
msg, _ := raw["message"].(string)
suggestion := suggestFix(int(statusCode))
return output.ErrorEnvelope(int(statusCode), msg, suggestion), &APIError{
StatusCode: int(statusCode),
Code: int(statusCode),
Message: msg,
}
bodyMsg, _ = raw["message"].(string)
} else if code, ok := raw["code"]; ok {
switch v := code.(type) {
case float64:
bodyCode = v
case int:
bodyCode = float64(v)
}
bodyMsg, _ = raw["msg"].(string)
if bodyMsg == "" {
bodyMsg, _ = raw["message"].(string)
}
}
if bodyCode != 0 && bodyCode != 200 && bodyCode != 201 && bodyCode != 204 && bodyCode != 1 {
suggestion := suggestFix(int(bodyCode))
return output.ErrorEnvelope(int(bodyCode), bodyMsg, suggestion), &APIError{
StatusCode: int(bodyCode),
Code: int(bodyCode),
Message: bodyMsg,
}
}
@ -197,6 +212,10 @@ func shouldAppendJSONSuffix(path string) bool {
return false
}
}
// Wiki open API endpoints do not use .json suffix
if len(parts) >= 3 && parts[0] == "wiki" && parts[1] == "open" {
return false
}
return true
}

View File

@ -169,6 +169,45 @@ func TestClientDoStatusError(t *testing.T) {
}
}
func TestClientDoGatewayCodeError(t *testing.T) {
// Gateway returns {"code":N, "msg":"..."} instead of {"status":N, "message":"..."}
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
w.Header().Set("Content-Type", "application/json")
w.Write([]byte(`{"code":400,"msg":"Bad Request"}`))
}))
defer server.Close()
c := &Client{HTTP: server.Client(), BaseURL: server.URL}
env, err := c.Do("GET", "/api/test", nil, nil)
if err == nil {
t.Fatal("expected error for code=400")
}
if env == nil {
t.Fatal("expected envelope for code error")
}
if env.OK {
t.Fatal("expected OK=false for code=400")
}
}
func TestClientDoGatewayCode201Success(t *testing.T) {
// Gateway returns code=201 with JSON string data — should be treated as success
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
w.Header().Set("Content-Type", "application/json")
w.Write([]byte(`{"code":201,"msg":"","data":"{\"title\":\"test\"}"}`))
}))
defer server.Close()
c := &Client{HTTP: server.Client(), BaseURL: server.URL}
env, err := c.Do("POST", "/api/test", map[string]string{"title": "test"}, nil)
if err != nil {
t.Fatalf("unexpected error for code=201: %v", err)
}
if !env.OK {
t.Fatal("expected OK=true for code=201")
}
}
func TestClientDoStatusZero(t *testing.T) {
// status=0, 200, 1 are treated as success
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
@ -525,3 +564,18 @@ func TestShouldAppendJSONSuffixSkipsExistingJSONPath(t *testing.T) {
t.Fatal("existing .json path should not get another suffix")
}
}
func TestShouldAppendJSONSuffixSkipsWikiOpenPaths(t *testing.T) {
paths := []string{
"/wiki/open/createWiki",
"/wiki/open/getWiki",
"/wiki/open/updateWiki",
"/wiki/open/deleteWiki",
"/wiki/open/wikiPages",
}
for _, p := range paths {
if shouldAppendJSONSuffix(p) {
t.Errorf("wiki/open path %q should not get .json suffix", p)
}
}
}

View File

@ -8,22 +8,25 @@ import (
)
const (
DefaultBaseURL = "https://www.gitlink.org.cn/api"
DefaultFormat = "table"
DefaultBaseURL = "https://www.gitlink.org.cn/api"
DefaultGatewayURL = "https://gateway.gitlink.org.cn/api"
DefaultFormat = "table"
)
type Config struct {
BaseURL string `yaml:"base_url"`
Format string `yaml:"default_format"`
Editor string `yaml:"editor,omitempty"`
Pager string `yaml:"pager,omitempty"`
Lang string `yaml:"lang,omitempty"`
BaseURL string `yaml:"base_url"`
GatewayURL string `yaml:"gateway_url"`
Format string `yaml:"default_format"`
Editor string `yaml:"editor,omitempty"`
Pager string `yaml:"pager,omitempty"`
Lang string `yaml:"lang,omitempty"`
}
func DefaultConfig() *Config {
return &Config{
BaseURL: DefaultBaseURL,
Format: DefaultFormat,
BaseURL: DefaultBaseURL,
GatewayURL: DefaultGatewayURL,
Format: DefaultFormat,
}
}
@ -54,6 +57,9 @@ func Load() (*Config, error) {
if cfg.BaseURL == "" {
cfg.BaseURL = DefaultBaseURL
}
if cfg.GatewayURL == "" {
cfg.GatewayURL = DefaultGatewayURL
}
if cfg.Format == "" {
cfg.Format = DefaultFormat
}
@ -80,6 +86,8 @@ func Get(key string) (string, error) {
switch key {
case "base_url":
return cfg.BaseURL, nil
case "gateway_url":
return cfg.GatewayURL, nil
case "default_format":
return cfg.Format, nil
case "editor":
@ -101,6 +109,8 @@ func Set(key, value string) error {
switch key {
case "base_url":
cfg.BaseURL = value
case "gateway_url":
cfg.GatewayURL = value
case "default_format":
cfg.Format = value
case "editor":

View File

@ -0,0 +1,59 @@
# scripts/research — 子赛题四·科研辅助算法层
本目录是子赛题四「应用 GitLink 辅助科研」的 **Python 工具代码**(赛题交付物之一)。
采用 **Go 出数据 + Python 做算法** 的分工:所有原始数据经现有 gitlink-cli25 个域)获取,
Python 负责知识图谱 / 协作匹配 / 可视化 / 复现性 / 报告等算法与产物生成。
## 文件
| 文件 | 场景 | 说明 |
|------|------|------|
| `gitlink_data.py` | 共享 | 调 gitlink-cli、解析 envelope、限速、分页、只读 health SQLite |
| `collect.py` | 共享 | 各命令的采集器 + 字段归一化login/repo 全名等) |
| `lineage.py` | S1 | 仓库级科研项目洞悉:提交/分支/PR/文档/实验代码演进 + 创新点 |
| `graph_build.py` | S2 | 科研知识图谱networkx节点/边模型 + 主题词典 + mermaid/DOT |
| `repro.py` | S3 | 合规与复现性license/密钥/依赖/复现清单 + 评分 |
| `match.py` | S4 | 科研协作智能匹配:学者画像 × 仓库缺口 TF-IDF + 余弦 |
| `report.py` | S5 | 科研进度智能跟踪与预警:周统计 + 里程碑 + 风险阈值 |
| `visual.py` | S6 | 科研成果可视化plotly 时间线/热力/饼/甘特/论文关联 |
| `templates/*.j2` | 全部 | jinja2 中文报告模板 |
| `test_*.py` | — | 单元测试(`pytest scripts/research/` |
## 安装
```bash
pip install -r scripts/research/requirements.txt
```
## 环境变量
| 变量 | 默认 | 说明 |
|------|------|------|
| `GITLINK_CLI` | `gitlink-cli` | CLI 可执行路径;本地 Windows 开发可设 `./gitlink-cli.exe` |
| `GITLINK_CLI_INTERVAL` | `0.6` | CLI 调用最小间隔(秒),防 API 限流 |
| `GITLINK_HEALTH_DB` | `~/.agents/skills/gitlink-health/data/gitlink_health.db` | health SQLite 路径(图谱/匹配读历史协作数据) |
## 数据来源
- **结构化数据**`repo/issue/pr/search/user/org/milestone/file/license` 等域,输出统一 envelope `{ok,data,meta}`
- **提交历史**Raw API `gitlink-cli api GET /{owner}/{repo}/commits`shortcuts 无 repo +commits路径已在 mindspore 仓库核验通过)。
- **历史协作**:直接只读 `health` SQLite`users/repos/issues/pulls/tags` 表)。
## 验证仓库
**主验证仓库:`mindspore-Ecosystem/mindspore`**(华为 MindSpore 深度学习框架镜像,真实科研级 AI 框架):
issues≈20346 / PR=9 / 贡献者=6 / star=31协作数据充足适合 S2/S4/S5同时是 Python 研究代码仓库,适合 S1/S3/S6。
S2 知识图谱为多仓库场景,按关键词 `search +repos` 跨仓库构建。
### 已核验响应形状mindspore 实测,供各场景脚本参考)
- `repo +info``default_branch`/`issues_count`/`pull_requests_count`/`contributor_users_count`/`size`/`clone_url`
- `repo +contributors[]``login`/`name`/`contributions`/`contribution_perc`/`email`
- `issue +list[]` → 标题 `subject`、时间 `created_at`、`status`、`priority_name`、`milestone_name`、`author`、`number`/`project_issues_index`
- `pr +list[]` → 标题 `title`、创建时间 `pr_created_unix`(秒)、`status`(0=open/1=merged/2=closed)、`reviewers`、`index`
- `api GET /commits[]``sha`/`message`/`timestamp`/`author.login`/`committer.login`
- `repo +languages``{"Python":"99.7%", ...}`
## 复现
每个场景配一个 `skills/<skill>/examples/<scenario>-workflow.sh`,串联 CLI → Python → 报告产物,
即赛题要求的「可复现执行脚本」。

262
scripts/research/collect.py Normal file
View File

@ -0,0 +1,262 @@
"""collect.py — 子赛题四共享数据采集器。
每个采集器是对 `gitlink-cli <domain> +<verb>` Raw API 的薄封装返回归一化后的
Python 对象dict/list所有 GitLink 操作经 gitlink-cli gitlink-shared 工具边界
形状说明GitLink 不同端点字段差异较大本模块只做尽力归一把不确定字段原样透传
具体字段解读留给各场景脚本lineage/graph/match/...并在其阶段用真实仓库验证后收敛
"""
from __future__ import annotations
from typing import Any, Iterable
import gitlink_data as gd
# ---------------------------------------------------------------------------
# 归一化小工具
# ---------------------------------------------------------------------------
def as_str(v: Any) -> str:
if v is None:
return ""
return str(v)
def as_int(v: Any, default: int = 0) -> int:
try:
return int(v)
except (TypeError, ValueError):
return default
def as_float(v: Any, default: float = 0.0) -> float:
try:
return float(v)
except (TypeError, ValueError):
return default
def login_of(obj: dict) -> str:
"""从用户/作者对象里尽量取出 loginGitLink 嵌套形式多变)。"""
if not isinstance(obj, dict):
return ""
for path_keys in (("login",), ("name",), ("username",),
("author", "login"), ("user", "login"),
("owner", "login"), ("author", "name")):
cur: Any = obj
ok = True
for k in path_keys:
if isinstance(cur, dict) and k in cur:
cur = cur[k]
else:
ok = False
break
if ok and isinstance(cur, str) and cur:
return cur
return ""
def repo_fullname(project: dict) -> str:
"""从 /projects 里的仓库对象取 'owner/identifier' 全名。"""
if not isinstance(project, dict):
return ""
identifier = project.get("identifier") or project.get("name") or project.get("repo_name") or ""
owner = login_of(project.get("author") or project.get("owner") or {}) or as_str(project.get("owner_login"))
if owner and identifier:
return f"{owner}/{identifier}"
return identifier
# ---------------------------------------------------------------------------
# 仓库级
# ---------------------------------------------------------------------------
def repo_info(owner: str, repo: str) -> dict:
return gd.run_data(["repo", "+info"], owner=owner, repo=repo) or {}
def readme(owner: str, repo: str, ref: str = "master") -> str:
data = gd.run_data(["repo", "+readme", "--ref", ref], owner=owner, repo=repo)
if isinstance(data, str):
return data
if isinstance(data, dict):
for k in ("content", "text", "readme", "markdown", "data"):
v = data.get(k)
if isinstance(v, str) and v:
return v
return ""
def tree(owner: str, repo: str, path: str = "", ref: str = "master") -> list:
flags = ["repo", "+tree", "--ref", ref]
if path:
flags += ["--path", path]
data = gd.run_data(flags, owner=owner, repo=repo)
return gd.first_list(data, ("entries", "trees", "files", "sub_entries"))
def languages(owner: str, repo: str) -> dict:
data = gd.run_data(["repo", "+languages"], owner=owner, repo=repo)
return data if isinstance(data, dict) else {}
def contributors(owner: str, repo: str, limit: int = 100) -> list:
"""贡献者列表。`repo +contributors` 无分页 flag见 shortcuts/repo/repo.go
故单次取全量limit 仅为兼容保留不传给 CLI"""
data = gd.run_data(["repo", "+contributors"], owner=owner, repo=repo)
return gd.first_list(data, ("contributors", "list"))
# ---------------------------------------------------------------------------
# Issue / PR / 里程碑(分页)
# ---------------------------------------------------------------------------
def issues(owner: str, repo: str, state: str = "all", max_pages: int = 10,
page_size: int = 50) -> list:
extra = ["--state", state] if state and state != "all" else []
return gd.paginate("issue", "list", owner=owner, repo=repo,
page_size=page_size, max_pages=max_pages, extra_flags=extra)
def prs(owner: str, repo: str, state: str = "all", max_pages: int = 10,
page_size: int = 50) -> list:
extra = ["--state", state] if state and state != "all" else []
# GitLink 把 PR 列表也放在 data["issues"] 键下(见 shortcuts/health/api.go
return gd.paginate("pr", "list", owner=owner, repo=repo,
list_keys=("issues", "pulls", "list"),
page_size=page_size, max_pages=max_pages, extra_flags=extra)
def issues_all(owner: str, repo: str, max_pages: int = 10,
page_size: int = 50) -> list:
"""全部 Issueopen + closed 合并去重)。
`issue +list` 默认只返 open open/closed 两个状态并集按 id 去重
每条保留真实 status 字段 S5/S6 做全量统计
"""
seen: dict = {}
for state in ("open", "closed"):
for iss in issues(owner, repo, state=state,
max_pages=max_pages, page_size=page_size):
key = iss.get("id") or iss.get("index")
if key is not None and key not in seen:
seen[key] = iss
return list(seen.values())
def prs_all(owner: str, repo: str, max_pages: int = 10,
page_size: int = 50) -> list:
"""全部 PRopen + merged + closed 合并去重)。
GitLink `pr +list` 状态过滤不可靠不同仓库行为不一有的默认只返 open
有的 --state 不生效返混合故取三个状态并集按 index 去重每条 PR 保留其
真实 status 字段'merged'/'open'/'closed'供上层分类仿 health 采集法
"""
seen: dict = {}
for state in ("open", "merged", "closed"):
for pr in prs(owner, repo, state=state,
max_pages=max_pages, page_size=page_size):
key = pr.get("index") or pr.get("id") or pr.get("number")
if key is None:
continue
if key not in seen:
seen[key] = pr
return list(seen.values())
def pr_detail(owner: str, repo: str, index: int) -> dict:
"""单个 PR 详情:`pr +view --id <index>`。
PR 列表 API 不返回文件改动数详情接口返回 `files_count` / `commits_count`
用于 S1 创新点识别的大规模重构判据
"""
return gd.run_data(["pr", "+view", "--id", str(index)], owner=owner, repo=repo) or {}
def milestones(owner: str, repo: str, state: str = "all") -> list:
extra = ["--status", state] if state and state != "all" else []
data = gd.run_data(["milestone", "+list", *extra, "--limit", "100"],
owner=owner, repo=repo)
return gd.first_list(data, ("milestones", "list"))
# ---------------------------------------------------------------------------
# 搜索 / 用户
# ---------------------------------------------------------------------------
def search_repos(keyword: str, limit: int = 20) -> list:
data = gd.run_data(["search", "+repos", "-k", keyword, "--limit", str(limit)])
return gd.first_list(data, ("projects", "repos"))
def search_users(keyword: str, limit: int = 20) -> list:
data = gd.run_data(["search", "+users", "-k", keyword, "--limit", str(limit)])
return gd.first_list(data, ("users", "list"))
def user_info(login: str) -> dict:
return gd.run_data(["user", "+info", "--login", login]) or {}
def user_repos(login: str, limit: int = 20) -> list:
"""某用户的公开仓库列表(`repo +list --user <login> --category all`)。
必须显式传 `--category all``repo +list` category 默认是 `manage`
"我管理的"用来列别人的仓库时会 404/返回 data.projects
每项含 identifier(仓库名)/language({id,name})/description
"""
data = gd.run_data(["repo", "+list", "--user", login,
"--category", "all", "--limit", str(limit)])
return gd.first_list(data, ("projects", "repos"))
# ---------------------------------------------------------------------------
# 提交历史Raw APIshortcuts 未提供 repo +commits
# ---------------------------------------------------------------------------
def commits(owner: str, repo: str, ref: str = "master", max_pages: int = 5,
page_size: int = 100) -> list:
"""通过 `api GET /{owner}/{repo}/commits` 分页取提交。
该端点确切路径/分页参数需在 Phase 0c 用真实仓库核验若不通
退化方案见 doc 注释按分支取或用 compare 端点
"""
out: list = []
for page in range(1, max_pages + 1):
data = gd.api("GET", f"/{owner}/{repo}/commits",
query=f"sha={ref}&page={page}&limit={page_size}",
owner=owner, repo=repo)
items = gd.first_list(data, ("commits", "list"))
if not items:
break
out.extend(items)
if len(items) < page_size:
break
return out
def file_text(owner: str, repo: str, path: str, ref: str = "master") -> str:
"""读取仓库内某文件的文本内容LICENSE / CI 配置 / lockfile 等)。
`file +get`sub_entries返回形如 {"entries": {"content": "...", "commit": {...}}}
"""
data = gd.run_data(["file", "+get", "--path", path, "--ref", ref],
owner=owner, repo=repo)
if isinstance(data, str):
return data
if isinstance(data, dict):
# 顶层直接带内容
for k in ("content", "text", "data"):
v = data.get(k)
if isinstance(v, str) and v:
return v
entries = data.get("entries")
# 形式一entries 是 dict内含 content 键GitLink 实测)
if isinstance(entries, dict):
v = entries.get("content") or entries.get("text")
if isinstance(v, str) and v:
return v
# 形式二entries 是 list[dict]
elif isinstance(entries, list) and entries and isinstance(entries[0], dict):
return as_str(entries[0].get("content") or entries[0].get("text"))
return ""

View File

@ -0,0 +1,196 @@
"""gitlink_data.py — 子赛题四数据访问共享层。
职责
1. 以子进程方式调用 gitlink-cli解析统一 envelope{ok,data,error,meta}
2. 内置限速避免触发 GitLink API 限流参考 shortcuts/health ~1.7 call/s
3. 列表分页累积
4. 直接只读访问 health SQLiteshortcuts/health/schema.sql S2 图谱 / S4 匹配取历史协作数据
设计原则本层只取数 + 解析不做任何业务算法算法在各场景脚本中实现
所有 GitLink 操作一律经 gitlink-cli绝不用 gh/glab skills/gitlink-shared/SKILL.md 工具边界
"""
from __future__ import annotations
import json
import os
import sqlite3
import subprocess
import sys
import time
from pathlib import Path
from typing import Any, Iterable
# ---------------------------------------------------------------------------
# 配置
# ---------------------------------------------------------------------------
# CLI 可执行路径:默认走 PATH 上的 gitlink-cli本地 Windows 开发可设 GITLINK_CLI=./gitlink-cli.exe
def cli_path() -> str:
return os.environ.get("GITLINK_CLI", "gitlink-cli")
# 最小调用间隔(秒),限制对 GitLink API 的请求频率。
MIN_INTERVAL = float(os.environ.get("GITLINK_CLI_INTERVAL", "0.6"))
_last_call = 0.0
def _throttle() -> None:
"""简单的全局令牌限速:两次调用间至少间隔 MIN_INTERVAL 秒。"""
global _last_call
now = time.time()
wait = MIN_INTERVAL - (now - _last_call)
if wait > 0:
time.sleep(wait)
_last_call = time.time()
def _warn(msg: str) -> None:
sys.stderr.write(f"[gitlink] {msg}\n")
# ---------------------------------------------------------------------------
# 核心:调用 CLI
# ---------------------------------------------------------------------------
def run(args: Iterable[str], owner: str | None = None, repo: str | None = None,
fmt: str = "json") -> dict[str, Any]:
"""调用 `gitlink-cli [global flags] args... --format fmt`,返回解析后的 envelope dict。
失败时返回 {"ok": False, "error": {...}}不抛异常便于调用方容错
"""
cmd = [cli_path()]
if owner:
cmd += ["--owner", owner]
if repo:
cmd += ["--repo", repo]
cmd += list(args)
if fmt:
cmd += ["--format", fmt]
_throttle()
try:
proc = subprocess.run(cmd, capture_output=True, text=True, timeout=120)
except FileNotFoundError:
_warn(f"gitlink-cli 未找到cmd={cmd[0]}),请设置 GITLINK_CLI 环境变量")
return {"ok": False, "error": {"message": f"gitlink-cli not found: {cmd[0]}"}}
except subprocess.TimeoutExpired:
_warn(f"调用超时:{' '.join(cmd)}")
return {"ok": False, "error": {"message": "timeout"}}
if proc.returncode != 0:
_warn(f"exit={proc.returncode} cmd={' '.join(cmd)}\n{proc.stderr.strip()}")
return {"ok": False, "error": {"message": (proc.stderr or proc.stdout).strip() or f"exit {proc.returncode}"}}
out = proc.stdout.strip()
if not out:
return {"ok": True, "data": None}
try:
return json.loads(out)
except json.JSONDecodeError:
# 非 JSON表格/原始文本),原样包裹返回
return {"ok": True, "data": out}
def run_data(args: Iterable[str], owner: str | None = None, repo: str | None = None,
fmt: str = "json") -> Any:
"""run() 的便捷封装:成功返回 data 字段,失败返回 None 并告警。"""
env = run(args, owner=owner, repo=repo, fmt=fmt)
if not env.get("ok"):
msg = (env.get("error") or {}).get("message", "unknown error")
_warn(f"{list(args)} -> {msg}")
return None
return env.get("data")
def api(method: str, path: str, query: str | None = None,
owner: str | None = None, repo: str | None = None) -> Any:
"""调用 Raw API`gitlink-cli api METHOD PATH --query ... --format json`。
用于 shortcuts 未覆盖的端点如仓库提交历史 GET /{owner}/{repo}/commits
"""
args = ["api", method, path]
if query:
args += ["--query", query]
return run_data(args, owner=owner, repo=repo)
# ---------------------------------------------------------------------------
# 分页 / 形状工具
# ---------------------------------------------------------------------------
# GitLink 各端点把列表放在不同键下;按优先级尝试这些键。
DEFAULT_LIST_KEYS = (
"projects", "repos", "issues", "pulls", "users", "list",
"contributors", "entries", "commits", "milestones", "tags",
)
def first_list(data: Any, keys: Iterable[str] = DEFAULT_LIST_KEYS) -> list:
"""从 envelope.data 里稳健地取出列表data 本身是列表则直接返回,否则尝试已知键。"""
if isinstance(data, list):
return data
if isinstance(data, dict):
for k in keys:
v = data.get(k)
if isinstance(v, list):
return v
# 兜底:唯一一个 list 值
list_vals = [v for v in data.values() if isinstance(v, list)]
if len(list_vals) == 1:
return list_vals[0]
return []
def total_count(data: Any) -> int | None:
if isinstance(data, dict):
for k in ("total_count", "totalCount", "count", "total"):
if isinstance(data.get(k), (int, float)):
return int(data[k])
return None
def paginate(domain: str, verb: str, list_keys=DEFAULT_LIST_KEYS,
page_size: int = 50, max_pages: int = 20,
owner: str | None = None, repo: str | None = None,
extra_flags: Iterable[str] = ()) -> list:
"""对一个 `gitlink-cli <domain> +<verb>` 列表命令做多页累积。
依赖命令支持 --page/--limit 两个 flagrepo/issue/pr/search/milestone 等均支持
"""
collected: list = []
for page in range(1, max_pages + 1):
flags = [f"+{verb}", "--page", str(page), "--limit", str(page_size), *extra_flags]
data = run_data([domain, *flags], owner=owner, repo=repo)
if data is None:
break
items = first_list(data, list_keys)
if not items:
break
collected.extend(items)
total = total_count(data)
if total is not None and len(collected) >= total:
break
if len(items) < page_size:
break
return collected
# ---------------------------------------------------------------------------
# health SQLite 只读访问
# ---------------------------------------------------------------------------
def health_db_path() -> str:
return os.environ.get(
"GITLINK_HEALTH_DB",
str(Path.home() / ".agents" / "skills" / "gitlink-health" / "data" / "gitlink_health.db"),
)
def open_health_db() -> sqlite3.Connection | None:
"""以只读方式打开 health SQLite文件不存在则返回 None调用方退化为纯 API 取数)。"""
p = health_db_path()
if not Path(p).exists():
return None
# mode=ro 防止误写modernc/sqlite 已开 WALPython 只读并发安全。
return sqlite3.connect(f"file:{p}?mode=ro", uri=True)

View File

@ -0,0 +1,658 @@
"""graph_build.py — S2 科研热点追踪与知识图谱。
输入一组科研关键词 GitLink 平台按关键词搜索相关仓库search +repos
对每个候选仓库取 repo_info / contributors / languages / README再用
networkx.MultiDiGraph 构建一张仓库学者主题科研知识图谱
节点
- repo : id = repo:owner/name (props: language/stars/forks/desc)
- scholar : id = scholar:login (来自 contributors过滤 bot/i-robot)
- topic : id = topic:x ( topics.py 词典抽取)
- contributes_to : scholar repo (weight = contribution_perc 解析为 0~1)
- owns : scholar repo ( author.login == contributor login)
- covers_topic : repo topic (weight = 出现次数 / max)
- collaborates_with : scholar scholar (共享同一 repo)
- related_to : topic topic (在同一 repo 共现)
取数建图严格分离build_graph() 只接收已经取好的 Python 数据结构
便于离线单测不联网不调 gitlink-clicollect() 负责在线取数
数据全部经 gitlink-cli 获取search +repos / repo +info / repo +contributors /
repo +languages / repo +readme
用法
python graph_build.py --keywords "deep learning,nlp" --repos-limit 20 --out ./out
python graph_build.py --keywords "knowledge graph" # 仅打印 JSON
"""
from __future__ import annotations
import argparse
import datetime as _dt
import json
import math
import os
import sys
import time
from collections import Counter, defaultdict
from typing import Any
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import collect as c # noqa: E402
import topics as T # noqa: E402
import networkx as nx # noqa: E402
# ---------------------------------------------------------------------------
# 小工具
# ---------------------------------------------------------------------------
BOT_LOGIN_HINTS = ("bot", "i-robot", "dependabot", "renovate", "semantic-release-bot")
def is_bot(login: str) -> bool:
"""识别明显机器人账号(不作为学者节点)。"""
if not login:
return True
low = login.lower()
return any(h in low for h in BOT_LOGIN_HINTS)
def parse_ratio(v: Any) -> float:
"""'1.18%' / '0.2' / 0.2 等统一解析为 0~1 比例(与 match.py 一致)。"""
if v is None:
return 0.0
s = str(v).strip()
pct = s.endswith("%")
if pct:
s = s[:-1]
try:
f = float(s)
except ValueError:
return 0.0
return f / 100.0 if (pct or f > 1.0) else f
def _topic_heat(descriptions: list[str], top: int = 10) -> list[dict]:
"""对全部仓库 description 跑 topics.topic_counter取 top 热度榜。"""
cnt = T.topic_counter(descriptions)
return [{"topic": t, "count": n} for t, n in cnt.most_common(top)]
# ---------------------------------------------------------------------------
# 热点追踪:飙升项目 + 活跃讨论(快照代理;真·增长率需定时轮询存历史)
# ---------------------------------------------------------------------------
def _to_epoch(v: Any) -> float:
"""把 GitLink 时间ISO 字符串或整数秒)解析为 epoch 秒,失败返回 0.0。"""
if v is None:
return 0.0
if isinstance(v, (int, float)):
return float(v) / 1000.0 if v > 1e12 else float(v)
s = str(v).strip()
if not s:
return 0.0
if s.isdigit():
f = float(s)
return f / 1000.0 if f > 1e12 else f
iso = s.replace("Z", "+00:00")
try:
return _dt.datetime.fromisoformat(iso).timestamp()
except (ValueError, TypeError):
import re
m = re.search(r"(\d{4})-(\d{2})-(\d{2})", s)
if m:
try:
return _dt.datetime(int(m.group(1)), int(m.group(2)), int(m.group(3))).timestamp()
except ValueError:
return 0.0
return 0.0
def _iso_day(epoch: float) -> str:
if not epoch:
return ""
try:
return _dt.datetime.utcfromtimestamp(epoch).strftime("%Y-%m-%d")
except (OSError, ValueError, OverflowError):
return ""
def compute_trending(repos: list[dict], top: int = 15) -> list[dict]:
"""对搜索到的仓库按「热度分数」排序,作为飙升/热门项目代理。
分数 = star + fork×2 + 近期更新加成并给日均增星(velocity)作为
day-1 可用的趋势代理·增长率需定时轮询存历史快照 trends 表规划
"""
now = time.time()
scored: list[dict] = []
for r in repos:
fullname = r.get("fullname") or c.repo_fullname(r)
if not fullname:
continue
stars = c.as_int(r.get("praises_count"))
forks = c.as_int(r.get("forked_count"))
updated = _to_epoch(r.get("updated_at") or r.get("time")
or r.get("updated_on") or r.get("created_at"))
age_days = max(1.0, (now - updated) / 86400.0) if updated else 99999.0
recency = max(0.0, 60.0 - age_days) # 60 天内更新有加成
score = stars + forks * 2 + recency
velocity = round(stars / age_days, 3) if age_days < 99990 else 0.0
lang_obj = r.get("language")
lang = lang_obj.get("name") if isinstance(lang_obj, dict) else c.as_str(lang_obj)
scored.append({
"repo": fullname,
"description": c.as_str(r.get("description"))[:140],
"language": lang or "",
"stars": stars,
"forks": forks,
"updated": _iso_day(updated),
"velocity": velocity,
"score": round(score, 1),
})
scored.sort(key=lambda x: -x["score"])
return scored[:top]
def compute_active(issues_map: dict[str, list], prs_map: dict[str, list],
top: int = 12) -> list[dict]:
"""跨仓库取评论/日志最多的 Issue / PR作为「活跃讨论」信号。"""
items: list[dict] = []
for fullname, issues in issues_map.items():
for iss in issues or []:
if not isinstance(iss, dict):
continue
jc = c.as_int(iss.get("journals_count") or iss.get("comments_count"))
items.append({
"repo": fullname, "type": "issue",
"number": iss.get("index") or iss.get("number") or iss.get("id"),
"title": c.as_str(iss.get("subject") or iss.get("title"))[:120],
"comments": jc,
"state": c.as_str(iss.get("status") or iss.get("issue_status") or "open"),
})
for fullname, prs in prs_map.items():
for pr in prs or []:
if not isinstance(pr, dict):
continue
jc = c.as_int(pr.get("journals_count") or pr.get("comments_count"))
items.append({
"repo": fullname, "type": "pr",
"number": pr.get("index") or pr.get("number") or pr.get("id"),
"title": c.as_str(pr.get("title"))[:120],
"comments": jc,
"state": c.as_str(pr.get("status") or "open"),
})
items.sort(key=lambda x: -x["comments"])
# 优先有评论的;不足则按已有顺序补齐
commented = [it for it in items if it["comments"] > 0]
return (commented or items)[:top]
# ---------------------------------------------------------------------------
# 建图(纯函数:不联网,只吃已取数据,便于单测)
# ---------------------------------------------------------------------------
def build_graph(repos: list[dict],
contributors_map: dict[str, list[dict]],
languages_map: dict[str, dict],
readmes: dict[str, str],
keywords: list[str] | None = None) -> dict[str, Any]:
"""构建科研知识图谱。
参数全部为已取好 Python 数据不联网
repos : list[dict]每个元素至少含 fullname repo_info 字段
identifier, author.login, description, language.name,
praises_count, forked_count
contributors_map : {fullname: [contributor, ...]}contributor 至少含
login + contribution_perc
languages_map : {fullname: {"Python": "99.7%", ...}}
readmes : {fullname: readme 文本已截断到前 4000 字符}
返回结果 dict graph.json 结构一致
"""
G = nx.MultiDiGraph()
repo_nodes: dict[str, dict] = {}
scholar_repos: dict[str, set[str]] = defaultdict(set)
repo_topics: dict[str, Counter] = {}
descriptions: list[str] = []
max_topic_count = 1 # 用于 covers_topic 归一化(防除零)
# ---- 1) 仓库 + 主题节点 ----
for r in repos:
fullname = r.get("fullname") or c.repo_fullname(r)
if not fullname:
continue
desc = c.as_str(r.get("description"))
lang_name = ""
lang_obj = r.get("language")
if isinstance(lang_obj, dict):
lang_name = c.as_str(lang_obj.get("name"))
stars = c.as_int(r.get("praises_count"))
forks = c.as_int(r.get("forked_count"))
readme = c.as_str(readmes.get(fullname))[:4000]
descriptions.append(desc)
repo_id = f"repo:{fullname}"
props = {
"language": lang_name,
"stars": stars,
"forks": forks,
"description": desc,
"readme_head": readme[:200],
}
G.add_node(repo_id, type="repo", label=fullname, **props)
repo_nodes[repo_id] = {"label": fullname, **props}
# 抽取主题description + readme
tps = T.extract_topics(desc + " " + readme)
cnt: Counter = Counter()
for tp in tps:
cnt[tp] += 1
repo_topics[repo_id] = cnt
if cnt:
max_topic_count = max(max_topic_count, max(cnt.values()))
# ---- 2) 主题节点 + covers_topic 边 ----
topic_repos: dict[str, set[str]] = defaultdict(set)
repo_topic_pairs: dict[str, list[str]] = {} # repo_id -> [topic_id]
for repo_id, cnt in repo_topics.items():
pairs: list[str] = []
for tp, n in cnt.items():
topic_id = f"topic:{tp}"
if topic_id not in G:
G.add_node(topic_id, type="topic", label=tp, count=0,
language="", stars=0, forks=0, description="")
# 累计该主题被多少仓库覆盖
G.nodes[topic_id]["count"] += 1
weight = n / max_topic_count if max_topic_count else 0.0
G.add_edge(repo_id, topic_id, type="covers_topic",
weight=round(weight, 4))
topic_repos[tp].add(repo_id)
pairs.append(topic_id)
repo_topic_pairs[repo_id] = pairs
# ---- 3) 学者节点 + contributes_to / owns ----
for fullname, contribs in contributors_map.items():
repo_id = f"repo:{fullname}"
if repo_id not in G:
continue
owner_login = ""
# 从 repos 列表里取该仓库 author.login判定 owns
for r in repos:
if (r.get("fullname") or c.repo_fullname(r)) == fullname:
owner_login = c.login_of(r.get("author") or {})
break
for contrib in contribs:
login = c.login_of(contrib)
if is_bot(login):
continue
scholar_id = f"scholar:{login}"
if scholar_id not in G:
G.add_node(scholar_id, type="scholar", label=login,
language="", stars=0, forks=0, description="")
weight = parse_ratio(contrib.get("contribution_perc"))
G.add_edge(scholar_id, repo_id, type="contributes_to",
weight=round(weight, 4))
if login == owner_login:
G.add_edge(scholar_id, repo_id, type="owns", weight=1.0)
scholar_repos[login].add(repo_id)
# ---- 4) collaborates_with共享同一 repo 的两两学者)----
for contribs in contributors_map.values():
logins = [c.login_of(x) for x in contribs if not is_bot(c.login_of(x))]
logins = sorted(set(logins))
if len(logins) < 2:
continue
for i in range(len(logins)):
for j in range(i + 1, len(logins)):
a = f"scholar:{logins[i]}"
b = f"scholar:{logins[j]}"
# 双向无向语义MultiDiGraph 用两条边近似)
G.add_edge(a, b, type="collaborates_with", weight=1.0)
G.add_edge(b, a, type="collaborates_with", weight=1.0)
# ---- 5) related_to同一 repo 内共现的两两主题)----
for repo_id, tids in repo_topic_pairs.items():
for i in range(len(tids)):
for j in range(i + 1, len(tids)):
a, b = tids[i], tids[j]
G.add_edge(a, b, type="related_to", weight=1.0)
G.add_edge(b, a, type="related_to", weight=1.0)
# ---- 6) 导出 ----
nodes_out = []
for nid, attrs in G.nodes(data=True):
nodes_out.append({
"id": nid,
"type": attrs.get("type", ""),
"label": attrs.get("label", nid),
"props": {k: v for k, v in attrs.items()
if k not in ("type", "label")},
})
edges_out = []
for u, v, attrs in G.edges(data=True):
edges_out.append({
"source": u,
"target": v,
"type": attrs.get("type", ""),
"weight": attrs.get("weight", 1.0),
})
# 核心学者:按出现 repo 数排序
core_scholars = sorted(
({"login": lg, "repo_count": len(rs)} for lg, rs in scholar_repos.items()),
key=lambda x: (-x["repo_count"], x["login"]),
)[:15]
# 核心团队:仅统计「组织」类型(非个人 User)的仓库拥有者,按拥有仓库数排序。
# 个人账号不算团队author.type 区分 User / Organization
owner_count: dict[str, int] = {}
owner_is_org: dict[str, bool] = {}
for r in repos:
author = r.get("author") or {}
lg = c.login_of(author)
if not lg:
continue
owner_count[lg] = owner_count.get(lg, 0) + 1
tp = str(author.get("type", "")).lower()
if tp and tp not in ("user", ""):
owner_is_org[lg] = True
owner_logins = sorted(
[{"login": lg, "repo_count": cnt, "type": "organization"}
for lg, cnt in owner_count.items() if owner_is_org.get(lg)],
key=lambda x: (-x["repo_count"], x["login"]),
)
# 主题热度:取所有 description 的 top10含图谱里实际命中的 count
heat = _topic_heat(descriptions, top=10)
return {
"scenario": "S2_research_knowledge_graph",
"keywords": list(keywords or []),
"nodes": nodes_out,
"edges": edges_out,
"core_scholars": core_scholars,
"core_teams": owner_logins,
"topic_heat": heat,
"meta": {
"keywords": list(keywords or []),
"repo_count": len(repo_nodes),
"node_count": G.number_of_nodes(),
"edge_count": G.number_of_edges(),
"scholar_count": sum(1 for n in nodes_out if n["type"] == "scholar"),
"topic_count": sum(1 for n in nodes_out if n["type"] == "topic"),
},
}
# ---------------------------------------------------------------------------
# 取数(在线:调 gitlink-cli
# ---------------------------------------------------------------------------
def collect(keywords: list[str], repos_limit: int = 20) -> dict[str, Any]:
"""按关键词搜索仓库并取其 info/contributors/languages/readme返回原始数据。
build_graph() 解耦本函数可被替换为 mock单测里直接构造数据喂 build_graph
"""
seen: dict[str, dict] = {} # fullname -> 归一化的 repo dict
for kw in keywords:
for r in c.search_repos(kw, limit=repos_limit):
fullname = c.repo_fullname(r)
if not fullname or fullname in seen:
continue
seen[fullname] = _normalize_search_hit(r, fullname)
if len(seen) >= repos_limit:
break
repos = list(seen.values())[:repos_limit]
contributors_map: dict[str, list[dict]] = {}
languages_map: dict[str, dict] = {}
readmes: dict[str, str] = {}
issues_map: dict[str, list[dict]] = {}
prs_map: dict[str, list[dict]] = {}
for r in repos:
fullname = r["fullname"]
owner, _, name = fullname.partition("/")
# 用仓库完整 info 覆盖搜索结果的稀疏字段
info = c.repo_info(owner, name)
if info:
r["description"] = c.as_str(info.get("description")) or r.get("description", "")
r["praises_count"] = c.as_int(info.get("praises_count") or info.get("watchers_count"))
r["forked_count"] = c.as_int(info.get("forked_count"))
# 更新时间(飙升/热度排序用GitLink 字段名兜底多个)
r["updated_at"] = (info.get("updated_at") or info.get("time")
or info.get("updated_on") or r.get("updated_at"))
if info.get("language") and isinstance(info["language"], dict):
r["language"] = info["language"]
contributors_map[fullname] = c.contributors(owner, name, limit=100)
languages_map[fullname] = c.languages(owner, name)
readmes[fullname] = c.readme(owner, name)[:4000]
# 活跃讨论:开放的 Issue / PR评论多的=热讨论)
issues_map[fullname] = c.issues(owner, name, state="open", max_pages=1, page_size=50)
prs_map[fullname] = c.prs(owner, name, state="open", max_pages=1, page_size=50)
return {"repos": repos, "contributors_map": contributors_map,
"languages_map": languages_map, "readmes": readmes,
"issues_map": issues_map, "prs_map": prs_map}
def _normalize_search_hit(r: dict, fullname: str) -> dict:
"""把 search_repos 返回项归一化为 build_graph 期望的形状。"""
lang_obj = r.get("language")
if not isinstance(lang_obj, dict):
lang_obj = {"name": c.as_str(lang_obj)}
return {
"fullname": fullname,
"identifier": r.get("identifier", fullname.split("/")[-1]),
"author": r.get("author") or {},
"description": c.as_str(r.get("description")),
"language": lang_obj,
"praises_count": c.as_int(r.get("praises_count")),
"forked_count": c.as_int(r.get("forked_count")),
"forked_from_project_id": r.get("forked_from_project_id"),
}
# ---------------------------------------------------------------------------
# 渲染Mermaid / DOT / Markdown 报告
# ---------------------------------------------------------------------------
_NODE_LIMIT = 40 # 防止 Mermaid 爆炸
_NODE_STYLE = {
"repo": ("repoNode", "#4C78A8"),
"scholar": ("scholarNode", "#F58518"),
"topic": ("topicNode", "#54A24B"),
}
def _safe_id(nid: str) -> str:
"""Mermaid/DOT 节点 id 用安全字符(去冒号斜杠)。"""
return nid.replace(":", "_").replace("/", "_").replace("-", "_")
def render_mermaid(result: dict[str, Any], node_limit: int = _NODE_LIMIT) -> str:
"""渲染前 ~40 节点的 Mermaid graph TD带 classDef 着色)。"""
lines = ["```mermaid", "graph TD"]
# 类定义
for t, (cls, color) in _NODE_STYLE.items():
lines.append(f" classDef {cls} fill:{color},stroke:#333,color:#fff;")
nodes = result.get("nodes", [])
edges = result.get("edges", [])
# 取前 node_limit 个节点repo 优先,再 scholar再 topic
type_order = {"repo": 0, "scholar": 1, "topic": 2}
ordered = sorted(nodes, key=lambda n: (type_order.get(n["type"], 9), n["id"]))
picked = ordered[:node_limit]
picked_ids = {n["id"] for n in picked}
label_map: dict[str, str] = {}
for n in picked:
sid = _safe_id(n["id"])
label = n["label"].replace('"', "'")
lines.append(f' {sid}["{label}"]')
label_map[n["id"]] = sid
cls = _NODE_STYLE.get(n["type"], ("", ""))[0]
if cls:
lines.append(f" class {sid} {cls};")
# 只画两端都在 picked 内的边;去重(同源同目标同类只画一条)
seen_edge: set[tuple] = set()
for e in edges:
if e["source"] not in picked_ids or e["target"] not in picked_ids:
continue
key = (e["source"], e["target"], e["type"])
if key in seen_edge:
continue
seen_edge.add(key)
a = label_map[e["source"]]
b = label_map[e["target"]]
w = e.get("weight", 1.0)
et = e["type"]
# 不同边类型用不同箭头标签
lines.append(f' {a} -- "{et}({w:.2f})" --> {b}')
lines.append("```")
return "\n".join(lines)
def render_dot(result: dict[str, Any], node_limit: int = _NODE_LIMIT) -> str:
"""渲染 Graphviz DOT 字符串(带节点着色)。"""
lines = ["digraph G {", ' rankdir=LR;',
' graph [fontname="Helvetica"];',
' node [fontname="Helvetica", style="filled"];',
' edge [fontname="Helvetica"];']
type_order = {"repo": 0, "scholar": 1, "topic": 2}
nodes = result.get("nodes", [])
edges = result.get("edges", [])
ordered = sorted(nodes, key=lambda n: (type_order.get(n["type"], 9), n["id"]))
picked = ordered[:node_limit]
picked_ids = {n["id"] for n in picked}
label_map: dict[str, str] = {}
for n in picked:
sid = _safe_id(n["id"])
label = n["label"].replace('"', "'")
color = _NODE_STYLE.get(n["type"], ("", "#CCCCCC"))[1]
lines.append(f' {sid} [label="{label}", fillcolor="{color}"];')
label_map[n["id"]] = sid
for e in edges:
if e["source"] not in picked_ids or e["target"] not in picked_ids:
continue
a = label_map[e["source"]]
b = label_map[e["target"]]
lines.append(f' {a} -> {b} [label="{e["type"]}"];')
lines.append("}")
return "\n".join(lines)
def render_report(result: dict[str, Any]) -> str:
meta = result["meta"]
heat = result.get("topic_heat", [])
scholars = result.get("core_scholars", [])
teams = result.get("core_teams", [])
lines = [
"# 科研热点追踪与知识图谱报告\n",
f"> 场景 S2 · 子赛题四「应用 GitLink 辅助科研」\n",
f"**关键词**: {', '.join(result.get('keywords') or []) or ''}\n",
"## 一、图谱概览\n",
f"- 仓库节点: **{meta['repo_count']}**",
f"- 学者节点: **{meta['scholar_count']}**",
f"- 主题节点: **{meta['topic_count']}**",
f"- 节点总数: **{meta['node_count']}**",
f"- 边总数: **{meta['edge_count']}**\n",
"## 二、主题热度榜(基于全部仓库 description\n",
"| 排名 | 主题 | 覆盖仓库数 |",
"|------|------|-----------|",
]
if heat:
for i, h in enumerate(heat, 1):
lines.append(f"| {i} | `{h['topic']}` | {h['count']} |")
else:
lines.append("| — | (未识别到明确主题) | — |")
lines += ["\n## 三、核心学者(按出现仓库数排序)\n",
"| 排名 | 学者 | 关联仓库数 |",
"|------|------|-----------|"]
if scholars:
for i, s in enumerate(scholars[:10], 1):
lines.append(f"| {i} | `{s['login']}` | {s['repo_count']} |")
else:
lines.append("| — | (未识别到学者) | — |")
lines += [f"\n## 四、核心团队(组织型仓库拥有者)\n"]
if teams:
lines += ["| 团队/组织 | 拥有仓库数 |", "|-----------|-----------|"]
for t in teams:
if isinstance(t, dict):
lines.append(f"| `{t.get('login')}` | {t.get('repo_count', 0)} |")
else:
lines.append(f"| `{t}` | — |")
else:
lines.append("(该批仓库均由个人账号拥有,无组织型团队)")
# 飙升/热门项目(热度分数排序;日均增星 velocity 作趋势代理)
trending = result.get("trending_repos", [])
lines += ["\n## 五、热门 / 飙升项目(热度排序)\n",
"| 仓库 | 语言 | ★ | ⑂ | 日均★ | 最近更新 |",
"|------|------|---:|---:|---:|----------|"]
if trending:
for t in trending[:10]:
lines.append(f"| `{t['repo']}` | {t['language'] or ''} | {t['stars']} | "
f"{t['forks']} | {t['velocity']} | {t['updated'] or ''} |")
else:
lines.append("| — | (未取到仓库) | | | | |")
# 活跃讨论
active = result.get("active_discussions", [])
lines += ["\n## 六、活跃讨论(评论最多的 Issue / PR\n",
"# | 类型 | 仓库 | 标题 | 评论 |", "|-|------|------|------|---:|"]
if active:
for i, a in enumerate(active[:10], 1):
lines.append(f"| {i} | {a['type']} | `{a['repo']}` | {a['title'][:50]} | {a['comments']} |")
else:
lines.append("| — | | | (暂无明显热讨论) | |")
lines.append("\n_配套产物graph.json结构化+ graph.mmdMermaid+ "
"graph.dotGraphviz DOT_\n")
return "\n".join(lines)
# ---------------------------------------------------------------------------
def main():
ap = argparse.ArgumentParser(description="S2 科研热点追踪与知识图谱")
ap.add_argument("--keywords", required=True,
help='逗号分隔的关键词,如 "deep learning,nlp"')
ap.add_argument("--repos-limit", type=int, default=20,
help="每关键词搜索后去重取 top N 仓库(默认 20")
ap.add_argument("--out", help="输出目录(写 graph.json/report.md/graph.mmd/graph.dot"
"省略则打印 JSON")
args = ap.parse_args()
keywords = [k.strip() for k in args.keywords.split(",") if k.strip()]
data = collect(keywords, repos_limit=args.repos_limit)
result = build_graph(data["repos"], data["contributors_map"],
data["languages_map"], data["readmes"], keywords=keywords)
# 热点追踪两翼:飙升项目 + 活跃讨论(图谱之外的"追踪"信号)
result["trending_repos"] = compute_trending(data["repos"])
result["active_discussions"] = compute_active(data["issues_map"], data["prs_map"])
if args.out:
os.makedirs(args.out, exist_ok=True)
with open(os.path.join(args.out, "graph.json"), "w", encoding="utf-8") as f:
json.dump(result, f, ensure_ascii=False, indent=2)
with open(os.path.join(args.out, "report.md"), "w", encoding="utf-8") as f:
f.write(render_report(result))
with open(os.path.join(args.out, "graph.mmd"), "w", encoding="utf-8") as f:
f.write(render_mermaid(result))
with open(os.path.join(args.out, "graph.dot"), "w", encoding="utf-8") as f:
f.write(render_dot(result))
print(f"✓ S2 知识图谱完成 → {args.out}/graph.json | report.md | graph.mmd | graph.dot")
print(f" 节点: {result['meta']['node_count']} 边: {result['meta']['edge_count']} "
f"仓库: {result['meta']['repo_count']}")
heat = result["topic_heat"][:5]
print(f" Top 主题: {', '.join(h['topic'] for h in heat)}")
else:
print(json.dumps(result, ensure_ascii=False, indent=2))
if __name__ == "__main__":
main()

507
scripts/research/hotspot.py Normal file
View File

@ -0,0 +1,507 @@
"""hotspot.py — 科研热点追踪(全栈重构版)。
输入一组科研关键词 GitLink 平台按关键词搜索相关仓库对每个仓库
- 拉取 repo_infostars / forks / 更新时间 / 贡献者
- 拉取 issue + pr 列表活跃讨论按评论数排序
- description + readme 抽取主题标签
- 聚合学者/团队贡献网络
输出hotspot.json结构化数据+ report.md中文简报
取数与算法分离collect() 在线取数compute_*() 纯函数离线可测
用法
python hotspot.py --keywords "deep learning,机器学习" --repos-limit 12 --out ./out
python hotspot.py --keywords "computer vision" # 仅打印 JSON
"""
from __future__ import annotations
import argparse
import json
import os
import sys
import time
from collections import Counter, defaultdict
from typing import Any
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import collect as c
import topics as T
# ---------------------------------------------------------------------------
# 小工具
# ---------------------------------------------------------------------------
def _now_ts() -> float:
return time.time()
def _days_ago(ts: float) -> int:
"""粗略计算距今多少天(非精确日历差,但足以排序)。"""
return max(0, int((_now_ts() - ts) / 86400))
# ---------------------------------------------------------------------------
# 热度评分(纯函数,离线可测)
# ---------------------------------------------------------------------------
def compute_trending_score(stars: int, forks: int, updated_days_ago: int) -> int:
"""仓库热度综合评分。
公式stars + forks × 2 + 近期更新加成30 天内加成最高
满分无上限用于仓库间横向排序
"""
base = stars + forks * 2
if updated_days_ago <= 7:
recency = 30
elif updated_days_ago <= 30:
recency = 20
elif updated_days_ago <= 90:
recency = 10
elif updated_days_ago <= 180:
recency = 5
else:
recency = 0
return base + recency
def compute_velocity(stars: int, updated_days_ago: int) -> float:
"""日均星标增速(近似值)。"""
if updated_days_ago <= 0:
updated_days_ago = 1
return round(stars / max(updated_days_ago, 1), 2)
# ---------------------------------------------------------------------------
# 在线取数
# ---------------------------------------------------------------------------
def collect(keywords: list[str], repos_limit: int = 12) -> dict:
"""在线取数:多关键词搜索 → 去重 → 仓库详情 → issue/PR → 贡献者 → README。
返回原始数据字典 compute() 消费
"""
# Step 1: 多关键词搜索
seen: dict[str, dict] = {}
for kw in keywords:
results = c.search_repos(kw, limit=repos_limit)
for proj in results:
full = c.repo_fullname(proj)
if not full:
continue
if full not in seen:
seen[full] = {
"fullname": full,
"matched_keywords": [],
"project": proj,
"info": None,
"issues": [],
"prs": [],
"contributors": [],
"readme": "",
"languages": {},
}
seen[full]["matched_keywords"].append(kw)
repos = list(seen.values())
# 先按 stars 粗排,截断到 repos_limit 再深度取数(节省 API 调用)
def _star_sort_key(r: dict) -> int:
p = r["project"]
return -(c.as_int(p.get("praises_count") or p.get("stars_count") or 0))
repos.sort(key=_star_sort_key)
repos = repos[:repos_limit]
# Step 2: 逐仓库取详情
for r in repos:
full = r["fullname"]
parts = full.split("/", 1)
if len(parts) != 2:
continue
owner, repo = parts[0], parts[1]
# repo_info
info = c.repo_info(owner, repo)
r["info"] = info if isinstance(info, dict) else {}
# issues取 open 状态,限制 5 页以控制耗时)
r["issues"] = c.issues(owner, repo, state="open", max_pages=3, page_size=30)
# PRs取 open 状态)
r["prs"] = c.prs(owner, repo, state="open", max_pages=3, page_size=30)
# contributors
r["contributors"] = c.contributors(owner, repo)
# readme前 4000 字符,仅用于主题抽取)
r["readme"] = (c.readme(owner, repo) or "")[:4000]
# languages
r["languages"] = c.languages(owner, repo)
return {"keywords": keywords, "repos": repos, "repos_limit": repos_limit}
# ---------------------------------------------------------------------------
# 离线计算(纯函数,不联网)
# ---------------------------------------------------------------------------
def compute(raw: dict) -> dict:
"""从 collect() 的原始数据计算出所有热点指标。
输入结构见 collect() 返回值输出为标准 hotspot.json 结构
"""
repos_raw = raw.get("repos") or []
# ------ trending_repos ------
trending: list[dict] = []
now = _now_ts()
for r in repos_raw:
info = r.get("info") or {}
proj = r.get("project") or {}
stars = c.as_int(
info.get("watchers_count")
or info.get("praises_count")
or proj.get("praises_count")
or 0
)
forks = c.as_int(
info.get("forked_count")
or proj.get("forked_count")
or 0
)
# 解析更新时间
update_str = (
info.get("full_last_update_time")
or info.get("last_update_time")
or proj.get("full_last_update_time")
or proj.get("last_update_time")
or ""
)
update_ts = _parse_ts(update_str)
days = _days_ago(update_ts)
score = compute_trending_score(stars, forks, days)
velocity = compute_velocity(stars, days)
language = ""
lang_obj = info.get("language") or proj.get("language")
if isinstance(lang_obj, dict):
language = lang_obj.get("name") or ""
elif isinstance(lang_obj, str):
language = lang_obj
trending.append({
"repo": r["fullname"],
"description": (
info.get("description")
or proj.get("description")
or ""
),
"language": language,
"stars": stars,
"forks": forks,
"score": score,
"velocity": velocity,
"matched_keywords": r.get("matched_keywords", []),
"updated": _fmt_ts(update_ts),
"contributors_count": c.as_int(info.get("contributor_users_count") or 0),
"releases_count": c.as_int(info.get("version_releases_count") or 0),
})
trending.sort(key=lambda x: -x["score"])
# ------ active_discussions ------
discussions: list[dict] = []
for r in repos_raw:
full = r["fullname"]
# issues
for iss in r.get("issues") or []:
comments = c.as_int(iss.get("comment_count") or iss.get("comments") or 0)
if comments > 0:
discussions.append({
"type": "issue",
"repo": full,
"title": iss.get("title") or "(无标题)",
"number": iss.get("index") or iss.get("number") or "",
"state": iss.get("state") or iss.get("status") or "open",
"comments": comments,
})
# PRs
for pr in r.get("prs") or []:
comments = c.as_int(pr.get("comment_count") or pr.get("comments") or 0)
if comments > 0:
discussions.append({
"type": "pr",
"repo": full,
"title": pr.get("title") or "(无标题)",
"number": pr.get("index") or pr.get("number") or "",
"state": pr.get("state") or pr.get("status") or "open",
"comments": comments,
})
discussions.sort(key=lambda x: -x["comments"])
# ------ topic_heat ------
# 对所有仓库的 description + readme 跑 topic_counter
texts = []
for r in repos_raw:
info = r.get("info") or {}
proj = r.get("project") or {}
desc = info.get("description") or proj.get("description") or ""
texts.append(desc)
readme = r.get("readme") or ""
if readme:
texts.append(readme)
heat = T.topic_counter(texts)
topic_heat = [
{"topic": topic, "count": count}
for topic, count in heat.most_common(10)
]
# ------ core_scholars / core_teams ------
scholar_repo_count: Counter = Counter()
# 记录每个 scholar 关联的仓库名
scholar_repos: dict[str, list[str]] = defaultdict(list)
for r in repos_raw:
full = r["fullname"]
for contrib in r.get("contributors") or []:
login = c.login_of(contrib)
if not login or _is_bot(login):
continue
scholar_repo_count[login] += 1
if full not in scholar_repos[login]:
scholar_repos[login].append(full)
core_scholars = [
{
"login": login,
"repo_count": count,
"repos": scholar_repos.get(login, []),
}
for login, count in scholar_repo_count.most_common(12)
]
# 核心团队:按 owner仓库第一段聚合
org_repo: Counter = Counter()
for r in repos_raw:
full = r["fullname"]
org = full.split("/")[0] if "/" in full else full
org_repo[org] += 1
core_teams = [
{"login": org, "repo_count": count}
for org, count in org_repo.most_common(8)
]
# ------ meta ------
total_issues = sum(len(r.get("issues") or []) for r in repos_raw)
total_prs = sum(len(r.get("prs") or []) for r in repos_raw)
return {
"scenario": "hotspot",
"keywords": raw.get("keywords") or [],
"trending_repos": trending,
"active_discussions": discussions,
"topic_heat": topic_heat,
"core_scholars": core_scholars,
"core_teams": core_teams,
"meta": {
"repo_count": len(repos_raw),
"issue_count": total_issues,
"pr_count": total_prs,
"scholar_count": len(scholar_repo_count),
"discussion_count": len(discussions),
"topic_count": len(topic_heat),
},
}
# ---------------------------------------------------------------------------
# 时间工具
# ---------------------------------------------------------------------------
def _parse_ts(v: Any) -> float:
"""把 GitLink 时间戳/字符串尽量解析为 Unix 浮点秒。"""
if v is None:
return 0.0
if isinstance(v, (int, float)):
if v > 1_000_000_000_000:
return v / 1000.0
return float(v)
s = str(v).strip()
if not s:
return 0.0
# ISO 8601 格式
for fmt in ("%Y-%m-%dT%H:%M:%S", "%Y-%m-%d %H:%M:%S",
"%Y-%m-%dT%H:%M:%SZ", "%Y-%m-%d"):
try:
return time.mktime(time.strptime(s[:19] if len(s) >= 19 else s, fmt))
except ValueError:
continue
return 0.0
def _fmt_ts(ts: float) -> str:
"""Unix 浮点秒 → 'YYYY-MM-DD' 字符串。"""
if ts <= 0:
return ""
try:
return time.strftime("%Y-%m-%d", time.localtime(ts))
except (ValueError, OSError):
return ""
_BOT_HINTS = ("bot", "i-robot", "dependabot", "renovate", "semantic-release-bot")
def _is_bot(login: str) -> bool:
low = login.lower()
return any(h in low for h in _BOT_HINTS)
# ---------------------------------------------------------------------------
# 报告生成
# ---------------------------------------------------------------------------
def _report_md(output: dict) -> str:
"""从热点 JSON 产出中文简报 Markdown。"""
kw = ", ".join(output.get("keywords") or [])
meta = output.get("meta") or {}
lines = [
f"# 🔬 科研热点追踪报告",
f"",
f"> 关键词:{kw}",
f"> 扫描时间:{_fmt_ts(_now_ts())}",
f"> 覆盖仓库:{meta.get('repo_count', 0)}"
f"· 讨论 {meta.get('discussion_count', 0)}"
f"· 主题 {meta.get('topic_count', 0)}"
f"· 学者 {meta.get('scholar_count', 0)}",
f"",
]
# 飙升项目 top 5
trending = output.get("trending_repos") or []
if trending:
lines.append("## 🔥 飙升项目 Top 5")
lines.append("")
lines.append("| # | 仓库 | 语言 | ★ Star | ⑂ Fork | 热度 | 更新 |")
lines.append("|---|------|------|--------|--------|------|------|")
for i, r in enumerate(trending[:5], 1):
lines.append(
f"| {i} | `{r['repo']}` | {r['language'] or ''} | "
f"{r['stars']} | {r['forks']} | {r['score']} | {r['updated']} |"
)
lines.append("")
# 活跃讨论 top 5
discussions = output.get("active_discussions") or []
if discussions:
lines.append("## 💬 活跃讨论 Top 5")
lines.append("")
for i, d in enumerate(discussions[:5], 1):
tp = "🐛 Issue" if d["type"] == "issue" else "🔀 PR"
lines.append(
f"{i}. {tp} [{d['repo']}] {d['title']} "
f"(#{d['number']} · {d['comments']} 💬)"
)
lines.append("")
# 热门主题
topic_heat = output.get("topic_heat") or []
if topic_heat:
lines.append("## 📊 热门主题")
lines.append("")
for t in topic_heat:
bar = "" * min(t["count"], 20)
lines.append(f"- **{t['topic']}** — {t['count']} 个仓库 {bar}")
lines.append("")
# 核心学者
scholars = output.get("core_scholars") or []
if scholars:
lines.append("## 👥 核心学者")
lines.append("")
for s in scholars[:5]:
lines.append(f"- **{s['login']}** — 关联 {s['repo_count']} 个仓库")
lines.append("")
# 核心团队
teams = output.get("core_teams") or []
if teams:
lines.append("## 🏛 活跃组织/团队")
lines.append("")
for t in teams:
lines.append(f"- **{t['login']}** — {t['repo_count']} 个仓库")
lines.append("")
lines.append("---")
lines.append("*由 gitlink-research-hotspot 自动生成*")
return "\n".join(lines)
# ---------------------------------------------------------------------------
# CLI 入口
# ---------------------------------------------------------------------------
def main() -> None:
ap = argparse.ArgumentParser(
description="科研热点追踪 — 多角度扫描 GitLink 科研仓库"
)
ap.add_argument(
"--keywords", "-k",
default="deep learning,机器学习",
help="搜索关键词,逗号分隔(默认 deep learning,机器学习)",
)
ap.add_argument(
"--repos-limit", type=int, default=12,
help="最大分析仓库数(默认 12",
)
ap.add_argument(
"--out", "-o",
default="",
help="输出目录(不传则仅打印 JSON 到 stdout",
)
args = ap.parse_args()
kw_list = [k.strip() for k in args.keywords.split(",") if k.strip()]
if not kw_list:
print(json.dumps({"ok": False, "error": "keywords required"}, ensure_ascii=False))
sys.exit(1)
# 取数
sys.stderr.write(f"[hotspot] 关键词: {kw_list} 上限: {args.repos_limit}\n")
sys.stderr.flush()
raw = collect(kw_list, args.repos_limit)
# 计算
sys.stderr.write(f"[hotspot] 仓库: {len(raw['repos'])} 计算热点…\n")
sys.stderr.flush()
output = compute(raw)
json_text = json.dumps(output, ensure_ascii=False, indent=2)
if args.out:
os.makedirs(args.out, exist_ok=True)
json_path = os.path.join(args.out, "hotspot.json")
with open(json_path, "w", encoding="utf-8") as f:
f.write(json_text)
report_path = os.path.join(args.out, "report.md")
with open(report_path, "w", encoding="utf-8") as f:
f.write(_report_md(output))
sys.stderr.write(
f"[hotspot] ✓ 完成 "
f"仓库={output['meta']['repo_count']} "
f"讨论={output['meta']['discussion_count']} "
f"主题={output['meta']['topic_count']} "
f"学者={output['meta']['scholar_count']}\n"
)
sys.stderr.write(f"[hotspot] 产物: {json_path}, {report_path}\n")
sys.stderr.flush()
else:
print(json_text)
if __name__ == "__main__":
main()

474
scripts/research/lineage.py Normal file
View File

@ -0,0 +1,474 @@
"""lineage.py — S1 仓库级科研项目洞悉。
对一个 GitLink 科研仓库做项目谱系lineage分析从默认分支提交时间线
合并 PR 的演进模式文档演进实验/评测文件组织关键创新点五个角度回答
这个科研项目是怎么一步步长成现在这样的值得引用/复现到哪一步
技术栈 = Go 出数据 gitlink-cli + collect.py 取数+ Python 做算法本文件
所有 GitLink 操作经 gitlink-cli禁止用 gh/glab gitlink-shared
用法
python lineage.py --owner mindspore-Ecosystem --repo mindspore --out ./out
python lineage.py --owner O --repo R # 仅打印 JSON
"""
from __future__ import annotations
import argparse
import json
import os
import re
import sys
from collections import Counter
from typing import Any
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import collect as c # noqa: E402
# ---------------------------------------------------------------------------
# 提交时间解析commit timestamp 可能是 ISO 字符串或整数秒)
# ---------------------------------------------------------------------------
def _to_epoch(v: Any) -> float:
"""把 GitLink commit 的 timestampISO 字符串或整数秒)统一解析为 epoch 秒。
失败返回 0.0
"""
if v is None:
return 0.0
if isinstance(v, (int, float)):
# 毫秒级时间戳兜底GitLink 多为秒)
return float(v) / 1000.0 if v > 1e12 else float(v)
s = str(v).strip()
if not s:
return 0.0
# 纯数字串
if s.isdigit():
f = float(s)
return f / 1000.0 if f > 1e12 else f
# ISO 8601'2024-05-01T08:00:00Z' / '2024-05-01 08:00:00'
s = s.replace("Z", "+00:00")
try:
import datetime as _dt
return _dt.datetime.fromisoformat(s).timestamp()
except (ValueError, TypeError):
# 退而求其次:抽首个 YYYY-MM-DD
m = re.search(r"(\d{4})-(\d{2})-(\d{2})", str(v))
if m:
try:
import datetime as _dt
return _dt.datetime(int(m.group(1)), int(m.group(2)),
int(m.group(3))).timestamp()
except ValueError:
return 0.0
return 0.0
def _iso_date(epoch: float) -> str:
"""epoch 秒 → 'YYYY-MM-DD' 字符串0 → '')。"""
if not epoch:
return ""
import datetime as _dt
try:
return _dt.datetime.utcfromtimestamp(epoch).strftime("%Y-%m-%d")
except (OSError, ValueError, OverflowError):
return ""
# ---------------------------------------------------------------------------
# 分类器:实验/评测文件 vs 文档文件
# ---------------------------------------------------------------------------
# 命中即判为「实验/评测/数据」类文件(科研产物信号)。
# 匹配「目录段或文件名」:以 experiment/benchmark/eval/test(s)/data 开头
# (后接 [\w-]* 续写,或紧接分隔符/串尾)。
_EXPERIMENT_RE = re.compile(
r"(^|[_/\-])(experiment[\w-]*|benchmark[\w-]*|eval[\w-]*|tests?|data)([/\-\._]|$)",
re.IGNORECASE,
)
def is_experiment_file(path: str) -> bool:
"""判断文件路径是否属于「实验/评测/数据」类(科研产物)。
命中 experiment*/benchmark*/eval*/tests?/data/ 任意一段作为目录名或文件名前缀
返回 True例如
- experiments/run.py, experiment_train.py, benchmark/eval.py
- tests/test_x.py, data/dataset.csv, src/benchmark_infer.py
"""
if not path:
return False
return bool(_EXPERIMENT_RE.search(path))
def is_doc_file(path: str) -> bool:
"""判断文件路径是否属于文档类:*.md任意位置或 docs/ 下任意文件。"""
if not path:
return False
low = path.lower()
if low.endswith(".md"):
return True
return low.startswith("docs/") or ("/docs/" in low)
# ---------------------------------------------------------------------------
# 分支地图(单分支简化:默认分支为唯一分支)
# ---------------------------------------------------------------------------
def build_branch_map(commits: list[dict], default_branch: str) -> list[dict]:
"""构建分支活跃度地图。本场景单分支简化:默认分支为唯一分支。
返回 [{name, commits, last_active, is_default}]
"""
n = len(commits) if isinstance(commits, list) else 0
last = 0.0
for cm in commits or []:
t = _to_epoch(cm.get("timestamp"))
if t and t > last:
last = t
return [{
"name": default_branch or "master",
"commits": n,
"last_active": _iso_date(last),
"is_default": True,
}]
# ---------------------------------------------------------------------------
# 提交时间线(按周/按日聚合)
# ---------------------------------------------------------------------------
def commit_timeline(commits: list[dict], bucket: str = "day") -> list[dict]:
"""把提交按日期聚合为时间线,返回按日期升序的 [{date, count}]。
bucket {'day','week'}week ISO - 表示
"""
counter: Counter = Counter()
for cm in commits or []:
t = _to_epoch(cm.get("timestamp"))
if not t:
continue
if bucket == "week":
import datetime as _dt
iso = _dt.datetime.utcfromtimestamp(t).isocalendar()
key = f"{iso[0]}-W{iso[1]:02d}"
else:
key = _iso_date(t)
counter[key] += 1
return [{"date": k, "count": counter[k]} for k in sorted(counter)]
# ---------------------------------------------------------------------------
# 合并 PR 的演进模式
# ---------------------------------------------------------------------------
def pr_merge_patterns(merged_prs: list[dict]) -> list[dict]:
"""从已合并 PR 抽取演进模式。
返回 [{number, title, status, merged_time, changed_files}]
changed_files pr 提供的 changed_files / changedFiles / additions/deletions 的近似
"""
out: list[dict] = []
for pr in merged_prs or []:
if not isinstance(pr, dict):
continue
# 时间:优先 merged 时间字段,否则 pr_created_unix
merged_t = (pr.get("pr_merged_unix") or pr.get("merged_at")
or pr.get("pr_updated_unix") or pr.get("pr_created_unix"))
# 文件改动数:优先详情接口的 files_count列表 API 不返回)
changed = (pr.get("files_count") or pr.get("changed_files")
or pr.get("changedFiles") or pr.get("file_nums") or 0)
# 兜底:用 additions/deletions 之和近似
if not changed:
changed = c.as_int(pr.get("additions"), 0) + c.as_int(pr.get("deletions"), 0)
# 标题只取首行 + 截断到 80 字(有些 PR 创建时把正文塞进了 title 字段)
raw_title = c.as_str(pr.get("title")).split("\n", 1)[0].strip()
title = raw_title[:80]
out.append({
"number": pr.get("index") or pr.get("number") or pr.get("id"),
"title": title,
"status": pr.get("status"),
"merged_time": _iso_date(_to_epoch(merged_t)),
"changed_files": c.as_int(changed),
})
# 按合并时间升序(空时间排末尾)
out.sort(key=lambda x: (x["merged_time"] == "", x["merged_time"]))
return out
# ---------------------------------------------------------------------------
# 文档演进docs/*.md 的近似最后修改信息)
# ---------------------------------------------------------------------------
def doc_evolution(tree_entries: list[dict]) -> list[dict]:
"""从仓库树抽取 docs/ 下的文档清单,近似其最后修改日期。
tree 来自 c.tree(owner, repo, path='docs')每条 entry 形如
{name, path, type, ...}最后修改日期在树端点通常不可得用文件名中的
日期或留空report 中标注近似
返回 [{file, last_date}]
"""
out: list[dict] = []
for e in tree_entries or []:
if not isinstance(e, dict):
continue
name = c.as_str(e.get("name") or e.get("path"))
path = c.as_str(e.get("path") or name)
if not is_doc_file(path):
continue
last = c.as_str(e.get("last_commit") or e.get("commit_date")
or e.get("date"))
if not last:
m = re.search(r"(\d{4})-(\d{2})-(\d{2})", path)
last = m.group(0) if m else ""
out.append({"file": name, "last_date": last})
out.sort(key=lambda x: (x["last_date"] == "", x["file"]))
return out
# ---------------------------------------------------------------------------
# 创新点识别(高影响合并)
# ---------------------------------------------------------------------------
def innovation_points(merged_prs: list[dict], commits: list[dict],
top: int = 8) -> list[dict]:
"""识别高影响合并作为项目的创新/里程碑点。
判据任一
- 改动文件数高>= 中位数的 1.5 或绝对值 >= 10 大规模重构/新特性
- 标题含里程碑关键词add/implement/feature/release/benchmark/... 特性引入
返回 [{description, evidence, category}]按影响度降序取前 top
"""
patterns = pr_merge_patterns(merged_prs)
if not patterns:
return []
files = [p["changed_files"] for p in patterns if p["changed_files"] > 0]
median = sorted(files)[len(files) // 2] if files else 0
MILESTONE_RE = re.compile(
r"(add|implement|support|feature|release|benchmark|refactor|"
r"experiment|dataset|train|inference|v\d+\.\d+)", re.IGNORECASE)
scored: list[tuple[float, dict]] = []
for p in patterns:
cf = p["changed_files"]
title = p["title"]
category = ""
impact = float(cf)
if cf >= 10 or (median and cf >= median * 1.5):
category = "大规模重构/新特性"
impact += 10
if MILESTONE_RE.search(title):
category = category or "特性引入"
impact += 5
if not category:
continue
evidence = (f"PR #{p['number']}{title[:48]}"
f"改动 {cf} 文件,合并于 {p['merged_time'] or '未知时间'}")
scored.append((impact, {
"description": title.strip() or f"PR #{p['number']}",
"evidence": evidence,
"category": category,
}))
scored.sort(key=lambda x: -x[0])
return [item for _, item in scored[:top]]
# ---------------------------------------------------------------------------
# 主流程:取数 + 算法
# ---------------------------------------------------------------------------
def lineage(owner: str, repo: str, branches_limit: int = 5) -> dict[str, Any]:
"""取数 + 分析,返回完整 lineage 结果 dict。"""
info = c.repo_info(owner, repo)
default_branch = info.get("default_branch") or "master"
commits = c.commits(owner, repo, ref=default_branch, max_pages=10, page_size=100)
# 已合并 PRstate=mergedcollect 透传)
merged_prs = c.prs(owner, repo, state="merged", max_pages=10, page_size=50)
# 列表 API 不返回文件改动数 → 逐个取详情补 files_count限速 + 上限 30 个,控 API 调用)
for pr in merged_prs[:30]:
idx = pr.get("index") or pr.get("number") or pr.get("id")
if idx is None:
continue
try:
det = c.pr_detail(owner, repo, int(idx))
except (TypeError, ValueError):
det = {}
if det:
if det.get("files_count") is not None:
pr["files_count"] = det.get("files_count")
if det.get("commits_count") is not None:
pr["commits_count"] = det.get("commits_count")
tree_root = c.tree(owner, repo, ref=default_branch)
docs_tree = c.tree(owner, repo, path="docs", ref=default_branch)
_readme = c.readme(owner, repo, ref=default_branch)
# 扫描整棵树,挑出实验/评测文件
exp_files: list[str] = []
all_tree = (tree_root or []) + (docs_tree or [])
for e in all_tree:
if not isinstance(e, dict):
continue
p = c.as_str(e.get("path") or e.get("name"))
if p and is_experiment_file(p):
exp_files.append(p)
timeline = commit_timeline(commits, bucket="day")
branch_map = build_branch_map(commits, default_branch)[:branches_limit]
pr_patterns = pr_merge_patterns(merged_prs)
docs = doc_evolution(docs_tree if docs_tree else tree_root)
innovations = innovation_points(merged_prs, commits)
return {
"scenario": "S1_repository_research_insight",
"repo": f"{owner}/{repo}",
"default_branch": default_branch,
"commit_timeline": timeline,
"branch_map": branch_map,
"pr_merge_patterns": pr_patterns,
"doc_evolution": docs,
"experiment_files": sorted(set(exp_files)),
"innovation_points": innovations,
"meta": {
"commit_count": len(commits),
"merged_pr_count": len(merged_prs),
"doc_count": len(docs),
"experiment_file_count": len(exp_files),
},
}
# ---------------------------------------------------------------------------
# 渲染Markdown 报告 + Mermaid gitGraph
# ---------------------------------------------------------------------------
def render_report(result: dict[str, Any]) -> str:
repo = result["repo"]
meta = result["meta"]
lines = [
f"# 仓库级科研项目洞悉报告 — {repo}\n",
f"> 场景 S1 · 子赛题四「应用 GitLink 辅助科研」· 项目谱系lineage分析\n",
"## 一、基础信息\n",
f"- **默认分支**: `{result['default_branch']}`",
f"- **采样提交**: {meta['commit_count']} 条(默认分支,最多 10×100",
f"- **已合并 PR**: {meta['merged_pr_count']}",
f"- **文档文件**: {meta['doc_count']}",
f"- **实验/评测文件**: {meta['experiment_file_count']}\n",
]
lines.append("## 二、提交活跃度时间线\n")
tl = result["commit_timeline"]
if tl:
peak = max(tl, key=lambda x: x["count"])
lines.append(f"- 时间跨度: {tl[0]['date']}{tl[-1]['date']}"
f"(共 {len(tl)} 个有提交的日期)")
lines.append(f"- 峰值: {peak['date']} 当日 {peak['count']} 次提交\n")
else:
lines.append("- (未取到提交时间线)\n")
lines.append("## 三、分支地图\n")
lines.append("| 分支 | 提交数 | 最后活跃 | 是否默认 |")
lines.append("|------|:------:|----------|:--------:|")
for b in result["branch_map"]:
lines.append(f"| `{b['name']}` | {b['commits']} | {b['last_active'] or ''} |"
f" {'' if b['is_default'] else ''} |")
lines.append("")
lines.append("## 四、合并 PR 演进模式(高影响合并预览)\n")
prs = result["pr_merge_patterns"]
if prs:
lines.append("| PR | 标题 | 改动文件 | 合并时间 |")
lines.append("|----|------|:--------:|----------|")
for p in prs[:10]:
lines.append(f"| #{p['number']} | {p['title']} | "
f"{p['changed_files']} | {p['merged_time'] or ''} |")
else:
lines.append("- (无已合并 PR")
lines.append("")
lines.append("## 五、创新/里程碑点\n")
inno = result["innovation_points"]
if inno:
for i, it in enumerate(inno, 1):
lines.append(f"{i}. **[{it['category']}]** {it['description']}")
lines.append(f" - 证据: {it['evidence']}")
else:
lines.append("- (未识别到明显高影响合并)")
lines.append("")
lines.append("## 六、文档演进docs/*\n")
docs = result["doc_evolution"]
if docs:
lines.append("| 文档 | 近似最后日期 |")
lines.append("|------|--------------|")
for d in docs[:15]:
lines.append(f"| {d['file']} | {d['last_date'] or ''} |")
else:
lines.append("- docs/ 下无文档或树不可得)")
lines.append("")
lines.append("## 七、实验/评测文件组织\n")
exps = result["experiment_files"]
if exps:
for p in exps[:20]:
lines.append(f"- `{p}`")
if len(exps) > 20:
lines.append(f"- ...(共 {len(exps)} 个,此处仅列前 20")
else:
lines.append("- (未在仓库树中识别到 experiment/benchmark/eval/test/data 目录)")
lines.append("")
return "\n".join(lines)
def render_mermaid(result: dict[str, Any]) -> str:
"""渲染 Mermaid gitGraph以默认分支为唯一分支按合并时间标里程碑 PR。"""
lines = ["```mermaid", "gitGraph"]
lines.append(f" commit id: \"{result['default_branch']} 起点\"")
prs = result["pr_merge_patterns"]
inno = result["innovation_points"]
inno_titles = {it["description"] for it in inno}
# 在时间线上穿插 commit / merge 标记(每 ~3 个 PR 一个 merge里程碑标 hotfix
step = 0
for p in prs:
lines.append(" commit")
step += 1
if step % 3 == 0:
tag = " 创新点" if any(p["title"].strip() == t for t in inno_titles) else ""
lines.append(f" commit id: \"#{p['number']}{tag}\" tag: \"{p['merged_time'] or 'PR'}\"")
lines.append("```")
return "\n".join(lines)
# ---------------------------------------------------------------------------
def main():
ap = argparse.ArgumentParser(description="S1 仓库级科研项目洞悉lineage 谱系分析)")
ap.add_argument("--owner", required=True)
ap.add_argument("--repo", required=True)
ap.add_argument("--branches-limit", type=int, default=5,
help="分支地图上限(本场景单分支简化,默认 5")
ap.add_argument("--out", help="输出目录(写 lineage.json/report.md/branch_graph.mmd"
"省略则打印 JSON")
args = ap.parse_args()
result = lineage(args.owner, args.repo, branches_limit=args.branches_limit)
if args.out:
os.makedirs(args.out, exist_ok=True)
with open(os.path.join(args.out, "lineage.json"), "w", encoding="utf-8") as f:
json.dump(result, f, ensure_ascii=False, indent=2)
with open(os.path.join(args.out, "report.md"), "w", encoding="utf-8") as f:
f.write(render_report(result))
with open(os.path.join(args.out, "branch_graph.mmd"), "w", encoding="utf-8") as f:
f.write(render_mermaid(result))
print(f"✓ S1 项目洞悉完成 → {args.out}/lineage.json | report.md | branch_graph.mmd")
print(f" 提交 {result['meta']['commit_count']} 条 | "
f"合并 PR {result['meta']['merged_pr_count']} 个 | "
f"创新点 {len(result['innovation_points'])}")
else:
print(json.dumps(result, ensure_ascii=False, indent=2))
if __name__ == "__main__":
main()

320
scripts/research/match.py Normal file
View File

@ -0,0 +1,320 @@
"""match.py — S4 科研协作智能匹配。
输入一个科研仓库分析其技术缺口未解决 Issue 的主题/语言开放 PR研究空缺
再从 GitLink 平台候选池本仓库贡献者 + 按缺口主题搜索到的用户主题向量 + 语言匹配 +
活跃度 + 协作开放度综合打分推荐最合适的跨团队/跨学者协作伙伴
数据全部经 gitlink-cli 获取issue +list / repo +contributors / repo +list --user / search +users
用法
python match.py --owner mindspore-Ecosystem --repo mindspore --top 10 --out ./out
python match.py --owner O --repo R --format json # 仅打印 JSON
"""
from __future__ import annotations
import argparse
import json
import math
import os
import sys
from collections import Counter
from typing import Any
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import collect as c # noqa: E402
import topics as T # noqa: E402
# 优先级 → 权重Issue 缺口信号加权)
PRIORITY_WEIGHT = {"urgent": 3, "high": 3, "紧急": 3, "": 3,
"normal": 2, "medium": 2, "普通": 2, "": 2,
"low": 1, "": 1}
# ---------------------------------------------------------------------------
# 向量与打分
# ---------------------------------------------------------------------------
def cosine(c1: dict[str, float], c2: dict[str, float]) -> float:
keys = set(c1) | set(c2)
dot = sum(c1.get(k, 0.0) * c2.get(k, 0.0) for k in keys)
n1 = math.sqrt(sum(v * v for v in c1.values()))
n2 = math.sqrt(sum(v * v for v in c2.values()))
return dot / (n1 * n2) if n1 and n2 else 0.0
def jaccard(a: list[str], b: list[str]) -> float:
sa, sb = set(a), set(b)
if not sa or not sb:
return 0.0
return len(sa & sb) / len(sa | sb)
def _priority_weight(issue: dict) -> float:
p = (issue.get("priority_name") or issue.get("priority") or "").lower()
for k, w in PRIORITY_WEIGHT.items():
if k in str(p):
return float(w)
return 1.0
def _parse_ratio(v: Any) -> float:
"""'1.18%' / '0.2' / 0.2 等统一解析为 0~1 比例。"""
if v is None:
return 0.0
s = str(v).strip()
pct = s.endswith("%")
if pct:
s = s[:-1]
try:
f = float(s)
except ValueError:
return 0.0
return f / 100.0 if (pct or f > 1.0) else f
# ---------------------------------------------------------------------------
# 缺口信号
# ---------------------------------------------------------------------------
def build_gap_signals(owner: str, repo: str, info: dict,
issue_sample: int = 100) -> tuple[Counter, list[str], list[dict]]:
"""返回 (缺口主题词频 Counter, 需求语言列表, 缺口信号明细)。"""
gap_topics: Counter = Counter()
gap_langs: set[str] = set()
signals: list[dict] = []
# 1) 仓库自身主题/语言(协作者应具备的基础方向)
desc = (info.get("description") or "") + " " + c.readme(owner, repo)[:4000]
for tp in T.extract_topics(desc):
gap_topics[tp] += 1
for lg in c.languages(owner, repo):
gap_langs.add(lg)
for lg in T.extract_languages(desc):
gap_langs.add(lg)
# 2) 未解决 Issue 的主题/语言(核心缺口)
open_issues = c.issues(owner, repo, state="open", max_pages=max(1, issue_sample // 50),
page_size=50)
for iss in open_issues[:issue_sample]:
text = (iss.get("subject") or iss.get("title") or "")
w = _priority_weight(iss)
for tp in T.extract_topics(text):
gap_topics[tp] += w
for lg in T.extract_languages(text):
gap_langs.add(lg)
# 取每条 issue 的首个主题作为该条的证据
tps = T.extract_topics(text)
if tps:
signals.append({"type": "unresolved_issue", "topic": tps[0],
"evidence": text[:80], "priority": iss.get("priority_name", "")})
# 3) 开放 PR 正在推进的方向(轻量加成)
for pr in c.prs(owner, repo, state="open", max_pages=1, page_size=30):
text = (pr.get("title") or "") + " " + (pr.get("body") or "")
for tp in T.extract_topics(text):
gap_topics[tp] += 0.5
needed_langs = sorted(gap_langs)
return gap_topics, needed_langs, signals
# ---------------------------------------------------------------------------
# 候选人画像与匹配
# ---------------------------------------------------------------------------
def candidate_pool(owner: str, repo: str, gap_topics: Counter, pool_cap: int) -> list[str]:
"""候选人 login 池:本仓库贡献者 + 按缺口主题搜到的外部用户。"""
seen: list[str] = []
seen_set: set[str] = set()
for contrib in c.contributors(owner, repo):
login = c.login_of(contrib)
# 过滤明显机器人账号
if login and login not in seen_set and "bot" not in login.lower() and login.lower() != "i-robot":
seen.append(login)
seen_set.add(login)
if len(seen) >= pool_cap:
return seen
# 取词频最高的若干主题,用其英文关键词搜外部用户
top_topics = [t for t, _ in gap_topics.most_common(5)]
eng_kw = {"nlp": "nlp", "deep_learning": "deep learning", "computer_vision": "cv",
"reinforcement_learning": "reinforcement learning",
"graph_learning": "gnn", "federated_learning": "federated",
"scientific_computing": "cuda", "data_mining": "machine learning",
"devops": "devops", "security": "security", "database": "database"}
for tp in top_topics:
kw = eng_kw.get(tp)
if not kw:
continue
for u in c.search_users(kw, limit=10):
login = c.login_of(u)
if login and login not in seen_set:
seen.append(login)
seen_set.add(login)
if len(seen) >= pool_cap:
break
return seen[:pool_cap]
def profile_candidate(login: str, repo_contribs: dict[str, dict]) -> dict[str, Any]:
"""构建候选人画像:主题向量 + 语言集合 + 活跃度 + 协作开放度。"""
repos = c.user_repos(login, limit=15)
texts = []
langs: set[str] = set()
fork_count = 0
for r in repos:
texts.append((r.get("description") or "") + " " + (r.get("identifier") or ""))
if r.get("language") and isinstance(r["language"], dict):
langs.add((r["language"].get("name") or "").lower())
if r.get("forked_from_project_id") or r.get("forked_count"):
fork_count += 1
topic_vec: Counter = Counter()
for t in texts:
for tp in T.extract_topics(t):
topic_vec[tp] += 1
for lg in T.extract_languages(" ".join(texts)):
langs.add(lg)
activity = 0.4
if login in repo_contribs:
# 本仓库贡献者 → 高活跃(贡献占比越高加成越大)
activity = 0.7 + 0.3 * min(_parse_ratio(repo_contribs[login].get("contribution_perc")), 1.0)
elif len(repos) >= 5:
activity = 0.7
elif repos:
activity = 0.4
collab = min(fork_count / 5.0, 1.0)
return {"topic_vec": dict(topic_vec), "langs": sorted(langs),
"activity": activity, "collab": collab, "repo_count": len(repos)}
def match(owner: str, repo: str, top: int = 10, pool_cap: int = 15,
issue_sample: int = 100) -> dict[str, Any]:
info = c.repo_info(owner, repo)
gap_topics, needed_langs, signals = build_gap_signals(owner, repo, info, issue_sample)
contribs_list = c.contributors(owner, repo)
repo_contribs = {c.login_of(x): x for x in contribs_list if c.login_of(x)}
# 缺口主题向量(与候选人主题向量同空间)
gap_vec = dict(gap_topics)
pool = candidate_pool(owner, repo, gap_topics, pool_cap)
scored = []
for login in pool:
prof = profile_candidate(login, repo_contribs)
topic_overlap = cosine(prof["topic_vec"], gap_vec)
lang_match = jaccard(prof["langs"], needed_langs) if needed_langs else 0.0
score = (0.45 * topic_overlap + 0.20 * lang_match
+ 0.20 * prof["activity"] + 0.15 * prof["collab"]) * 100
reasons: list[str] = []
overlap_topics = sorted(set(prof["topic_vec"]) & set(gap_vec),
key=lambda k: -prof["topic_vec"][k])
if overlap_topics:
reasons.append(f"覆盖缺口主题: {', '.join(overlap_topics[:4])}")
matched_langs = sorted(set(prof["langs"]) & set(needed_langs))
if matched_langs:
reasons.append(f"语言匹配: {', '.join(matched_langs[:4])}")
if login in repo_contribs:
reasons.append("本仓库活跃贡献者")
if prof["collab"] > 0:
reasons.append(f"协作开放度高(fork={int(prof['collab']*5)})")
activity_level = ("high" if prof["activity"] >= 0.7
else "medium" if prof["activity"] >= 0.4 else "low")
scored.append({
"login": login, "score": round(score, 1),
"topic_overlap": round(topic_overlap, 3),
"language_match": round(lang_match, 3),
"activity_level": activity_level,
"repo_languages": prof["langs"][:6],
"repo_count": prof["repo_count"],
"reasons": reasons or ["无明显主题/语言重叠"],
})
scored.sort(key=lambda x: -x["score"])
top_topics = [t for t, _ in gap_topics.most_common(8)]
return {
"scenario": "S4_collaboration_matching",
"repo": f"{owner}/{repo}",
"gap_topics": top_topics,
"needed_languages": needed_langs,
"gap_signals": signals[:30],
"candidates": scored[:top],
"meta": {"pool_size": len(pool), "issue_sample": issue_sample},
}
# ---------------------------------------------------------------------------
# 渲染Markdown 报告 + Mermaid 协作网络
# ---------------------------------------------------------------------------
def render_report(result: dict[str, Any]) -> str:
repo = result["repo"]
cands = result["candidates"]
lines = [
f"# 科研协作智能匹配报告 — {repo}\n",
f"> 场景 S4 · 子赛题四「应用 GitLink 辅助科研」\n",
"## 一、仓库技术缺口分析\n",
f"- **缺口主题**: {', '.join(result['gap_topics']) or '(未识别到明确主题)'}",
f"- **需求语言**: {', '.join(result['needed_languages']) or ''}",
f"- **缺口信号样本**: {len(result['gap_signals'])} 条未解决 Issue/PR 主题证据\n",
"| 缺口主题 | 证据Issue/PR | 优先级 |",
"|----------|------------------|--------|",
]
for s in result["gap_signals"][:8]:
lines.append(f"| {s['topic']} | {s['evidence']} | {s.get('priority','')} |")
lines += ["\n## 二、推荐协作伙伴(按综合匹配分排序)\n",
"| 排名 | 用户 | 匹配分 | 主题重叠 | 语言匹配 | 活跃度 | 匹配理由 |",
"|------|------|--------|----------|----------|--------|----------|"]
for i, m in enumerate(cands, 1):
lines.append(f"| {i} | `{m['login']}` | {m['score']} | {m['topic_overlap']} | "
f"{m['language_match']} | {m['activity_level']} | {'; '.join(m['reasons'][:2])} |")
lines.append(f"\n_候选池规模 {result['meta']['pool_size']}issue 采样 {result['meta']['issue_sample']}_\n")
return "\n".join(lines)
def render_mermaid(result: dict[str, Any]) -> str:
repo = result["repo"].replace("/", "_")
lines = ["```mermaid", "graph TD", f' R["{result["repo"]}<br/>(目标仓库)"]']
for i, m in enumerate(result["candidates"][:8], 1):
nid = f"C{i}"
lines.append(f' {nid}["{m["login"]}<br/>{m["score"]}"]')
# 边的粗细用文字标签近似
lines.append(f' R -- "{m["topic_overlap"]}" --> {nid}')
lines.append("```")
return "\n".join(lines)
# ---------------------------------------------------------------------------
def main():
ap = argparse.ArgumentParser(description="S4 科研协作智能匹配")
ap.add_argument("--owner", required=True)
ap.add_argument("--repo", required=True)
ap.add_argument("--top", type=int, default=10)
ap.add_argument("--pool", type=int, default=15, help="候选池上限")
ap.add_argument("--issue-sample", type=int, default=100)
ap.add_argument("--out", help="输出目录(写 match.json/report.md/network.mmd省略则打印 JSON")
args = ap.parse_args()
result = match(args.owner, args.repo, top=args.top, pool_cap=args.pool,
issue_sample=args.issue_sample)
if args.out:
os.makedirs(args.out, exist_ok=True)
with open(os.path.join(args.out, "match.json"), "w", encoding="utf-8") as f:
json.dump(result, f, ensure_ascii=False, indent=2)
with open(os.path.join(args.out, "report.md"), "w", encoding="utf-8") as f:
f.write(render_report(result))
with open(os.path.join(args.out, "network.mmd"), "w", encoding="utf-8") as f:
f.write(render_mermaid(result))
print(f"✓ S4 匹配完成 → {args.out}/match.json | report.md | network.mmd")
print(f" 缺口主题: {', '.join(result['gap_topics'])}")
print(f" Top 推荐: {', '.join(m['login']+'('+str(m['score'])+')' for m in result['candidates'][:5])}")
else:
print(json.dumps(result, ensure_ascii=False, indent=2))
if __name__ == "__main__":
main()

555
scripts/research/report.py Normal file
View File

@ -0,0 +1,555 @@
"""report.py — S5 科研进度智能跟踪与预警。
输入一个科研仓库统计本周 / 上周的提交IssuePR 活跃度结合里程碑进度
用阈值规则产出风险预警stale issue / stale PR / 逾期里程碑 / 低活跃 / bus_factor
并给出本周相对上周的 commit 趋势辅助科研负责人及时发现项目停滞 / 单点风险
数据全部经 gitlink-cli 获取commits / issue +list / pr +list / milestone +list /
repo +contributors算法为纯函数零第三方依赖单测以 mock 数据喂入
用法
python report.py --owner mindspore-Ecosystem --repo mindspore --out ./out
python report.py --owner O --repo R # 仅打印 JSON
"""
from __future__ import annotations
import argparse
import json
import os
import sys
from collections import Counter
from datetime import datetime, timedelta, timezone
from typing import Any
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import collect as c # noqa: E402
# ---------------------------------------------------------------------------
# 常量(阈值,便于单测覆盖)
# ---------------------------------------------------------------------------
STALE_ISSUE_DAYS = 30 # 开放且无活动 > 30 天 → stale issue
STALE_PR_DAYS = 14 # 开放且未 review > 14 天 → stale PR
STALE_PR_WINDOW_DAYS = 90 # 统计 stale PR 时回溯窗口(避免扫全量历史)
LOW_ACTIVITY_COMMITS = 3 # 本周提交 < 3 → 低活跃
BUS_FACTOR_RATIO = 0.5 # 单一贡献者占本周提交 > 50% → bus factor
UTC = timezone.utc
# ---------------------------------------------------------------------------
# 时间解析
# ---------------------------------------------------------------------------
def parse_time(s: Any) -> datetime | None:
"""把时间字段解析为带 UTC 时区的 datetime无法解析返回 None。
兼容两类输入
- ISO 字串 '2024-06-01T08:30:00+08:00' / '2024-06-01T08:30:00Z'
/ '2024-06-01 08:30:00' / '2024-06-01'
- 整数或整数字串秒级 Unix 时间戳 1717200000 / '1717200000'
"""
if s is None:
return None
if isinstance(s, (int, float)):
try:
return datetime.fromtimestamp(float(s), tz=UTC)
except (OverflowError, OSError, ValueError):
return None
if not isinstance(s, str):
return None
text = s.strip()
if not text:
return None
# 纯数字 → 当作 Unix 秒级时间戳
if text.lstrip("-").isdigit():
try:
return datetime.fromtimestamp(float(text), tz=UTC)
except (OverflowError, OSError, ValueError):
return None
# ISO 字串:统一以 Z → +00:00
iso = text.replace("Z", "+00:00")
try:
dt = datetime.fromisoformat(iso)
except ValueError:
# 尝试 'YYYY-MM-DD HH:MM:SS' / 'YYYY-MM-DD'
for fmt in ("%Y-%m-%d %H:%M:%S", "%Y-%m-%d"):
try:
dt = datetime.strptime(text, fmt)
break
except ValueError:
continue
else:
return None
if dt.tzinfo is None:
dt = dt.replace(tzinfo=UTC)
return dt.astimezone(UTC)
def in_window(dt: datetime | None, days: int, now: datetime) -> bool:
"""判断 dt 是否落在 [now - days, now] 区间内。None 视为不在窗口内。"""
if dt is None:
return False
if days < 0:
return False
now_utc = now.astimezone(UTC) if now.tzinfo else now.replace(tzinfo=UTC)
dt_utc = dt.astimezone(UTC) if dt.tzinfo else dt.replace(tzinfo=UTC)
return (now_utc - dt_utc) <= timedelta(days=days) and (now_utc - dt_utc) >= timedelta(0)
# ---------------------------------------------------------------------------
# 周统计
# ---------------------------------------------------------------------------
def _commit_time(item: dict) -> datetime | None:
for key in ("timestamp", "commit_time", "committed_date", "created_at"):
v = item.get(key)
dt = parse_time(v)
if dt is not None:
return dt
return None
def _issue_time(item: dict, prefer: tuple[str, ...]) -> datetime | None:
for key in prefer:
v = item.get(key)
dt = parse_time(v)
if dt is not None:
return dt
return None
def _issue_status(item: dict) -> str:
s = item.get("status")
if isinstance(s, str) and s:
return s.lower()
return ""
def _pr_status(item: dict) -> int:
"""PR 状态 → 0=open,1=merged,2=closed。
GitLink PR 列表 status 是字符串('merged'/'open'/'closed')
详情/health 可能是 pull_request_status 整数(0/1/2)两者兼容
"""
v = item.get("status", item.get("pull_request_status", 0))
if isinstance(v, str):
s = v.lower()
if "merged" in s:
return 1
if s in ("closed", "close", "reject", "rejected"):
return 2
return 0
try:
return int(v)
except (TypeError, ValueError):
return 0
def _author_login(item: dict, key: str = "author") -> str:
"""从 commit/issue/pr 的 author 子对象取 login退化用顶层 login/name。"""
obj = item.get(key)
if isinstance(obj, dict):
for k in ("login", "name", "username"):
if obj.get(k):
return str(obj[k])
for k in ("login", "name", "username", "committer_login"):
if item.get(k):
return str(item[k])
return ""
def _week_bounds(now: datetime) -> tuple[datetime, datetime, datetime, datetime]:
"""返回 (this_week_start, last_week_start, last_week_end, now)UTC
按自然天对齐本周 = [now-7d, now]上周 = [now-14d, now-7d)
"""
now_utc = now.astimezone(UTC) if now.tzinfo else now.replace(tzinfo=UTC)
this_start = now_utc - timedelta(days=7)
last_start = now_utc - timedelta(days=14)
last_end = this_start
return this_start, last_start, last_end, now_utc
def _commits_in(commits: list[dict], lo: datetime, hi: datetime) -> list[dict]:
out = []
for it in commits:
dt = _commit_time(it)
if dt is None:
continue
if lo <= dt < hi:
out.append(it)
return out
def _window_summary(commits: list, issues: list, prs: list,
contributors: list, lo: datetime, hi: datetime,
now: datetime) -> dict[str, Any]:
"""统计 [lo, hi) 区间内的活动摘要(供 week_stats 复用)。"""
w_commits = _commits_in(commits, lo, hi)
opened = closed = stale_open = 0
for iss in issues:
created = _issue_time(iss, ("created_at", "created_unix"))
if created is not None and lo <= created < hi:
opened += 1
st = _issue_status(iss)
if st in ("closed", "reject", "rejected"):
# 关闭时间优先 closed_at/journals_updated_at
closed_dt = _issue_time(iss, ("closed_at", "journals_updated_at", "updated_at"))
if closed_dt is not None and lo <= closed_dt < hi:
closed += 1
elif st in ("", "open", "opened"):
# stale 判定:开放且距今 > STALE_ISSUE_DAYS 无活动
last_dt = _issue_time(iss, ("journals_updated_at", "updated_at", "created_at"))
if last_dt is not None and (now - last_dt) > timedelta(days=STALE_ISSUE_DAYS):
stale_open += 1
pr_opened = pr_merged = pr_open_stale = 0
for pr in prs:
created = _issue_time(pr, ("pr_created_unix", "created_at", "created_unix"))
if created is None:
created = _commit_time(pr)
if created is not None and lo <= created < hi:
pr_opened += 1
st = _pr_status(pr)
if st == 1: # merged
merged_dt = _issue_time(pr, ("pr_merged_unix", "merged_at", "updated_at"))
if merged_dt is None:
merged_dt = created
if merged_dt is not None and lo <= merged_dt < hi:
pr_merged += 1
elif st == 0: # open
# stale PR开放且 created > STALE_PR_DAYS 天前、近 STALE_PR_WINDOW 天内
if created is not None and (now - created) > timedelta(days=STALE_PR_DAYS) \
and created > now - timedelta(days=STALE_PR_WINDOW_DAYS):
pr_open_stale += 1
# 本周活跃贡献者:去重 author logincommits
logins = [_author_login(it) for it in w_commits]
contrib_active = {lg for lg in logins if lg}
return {
"commits": len(w_commits),
"issues_opened": opened,
"issues_closed": closed,
"issues_stale": stale_open,
"prs_opened": pr_opened,
"prs_merged": pr_merged,
"prs_open_stale": pr_open_stale,
"contributors_active": len(contrib_active),
"contributors_active_logins": sorted(contrib_active),
}
def week_stats(commits: list, issues: list, prs: list,
contributors: list, now: datetime) -> dict[str, Any]:
"""统计本周 / 上周活跃度摘要。contributors 形参保留以兼容,活跃度以 commits 的作者为准。"""
this_start, last_start, last_end, now_utc = _week_bounds(now)
this = _window_summary(commits, issues, prs, contributors,
this_start, now_utc, now_utc)
last = _window_summary(commits, issues, prs, contributors,
last_start, last_end, now_utc)
return {
"this_week": this,
"last_week": last,
"window": {
"this_week_start": this_start.isoformat(),
"now": now_utc.isoformat(),
"last_week_start": last_start.isoformat(),
"last_week_end": last_end.isoformat(),
},
"total_contributors": len(contributors),
}
# ---------------------------------------------------------------------------
# 里程碑进度
# ---------------------------------------------------------------------------
def _milestone_due(item: dict) -> datetime | None:
for key in ("effective_date", "due_date", "deadline", "end_date"):
dt = parse_time(item.get(key))
if dt is not None:
return dt
return None
def milestone_progress(milestones: list, issues: list, now: datetime) -> list[dict]:
"""每个里程碑的 open/closed issue 数、完成率、是否逾期。
Issue Milestone 的关联字段优先用 milestone_name / milestone_id
"""
now_utc = now.astimezone(UTC) if now.tzinfo else now.replace(tzinfo=UTC)
by_key: dict[Any, dict[str, int]] = {}
for iss in issues:
ms_name = iss.get("milestone_name") or iss.get("milestone")
ms_id = iss.get("milestone_id") or iss.get("milestone_index")
key = ms_name if ms_name else (ms_id if ms_id is not None else None)
if key is None:
continue
bucket = by_key.setdefault(key, {"open": 0, "closed": 0})
st = _issue_status(iss)
if st in ("closed", "reject", "rejected"):
bucket["closed"] += 1
else:
bucket["open"] += 1
out: list[dict] = []
for ms in milestones:
name = ms.get("name") or ms.get("title") or "(未命名)"
key = name
counts = by_key.get(key, {"open": 0, "closed": 0})
total = counts["open"] + counts["closed"]
pct = round(100.0 * counts["closed"] / total, 1) if total else 0.0
due = _milestone_due(ms)
overdue = False
# 仅当里程碑未关闭 + 有 due_date 且 due < now → 逾期
ms_status = str(ms.get("status", "")).lower()
is_closed = ms_status in ("closed", "reject", "rejected", "done", "completed")
if due is not None and not is_closed and due < now_utc:
overdue = True
out.append({
"name": name,
"open": counts["open"],
"closed": counts["closed"],
"total": total,
"completion_pct": pct,
"due_date": due.isoformat() if due else None,
"overdue": overdue,
"status": ms.get("status", ""),
})
return out
# ---------------------------------------------------------------------------
# 趋势
# ---------------------------------------------------------------------------
def trend(this_week: dict, last_week: dict) -> dict[str, Any]:
"""本周 vs 上周 commit 增量与活跃度等级。"""
tc = this_week.get("commits", 0)
lc = last_week.get("commits", 0)
if lc == 0:
delta_pct = 100.0 if tc > 0 else 0.0
else:
delta_pct = round(100.0 * (tc - lc) / lc, 1)
if delta_pct > 10:
level = "increasing"
elif delta_pct < -10:
level = "decreasing"
else:
level = "stable"
return {"commit_delta_pct": delta_pct, "activity_level": level,
"this_week_commits": tc, "last_week_commits": lc}
# ---------------------------------------------------------------------------
# 风险预警
# ---------------------------------------------------------------------------
def risk_warnings(stats: dict, milestones: list, contributors: list,
commits: list | None = None, now: datetime | None = None) -> list[dict]:
"""阈值规则 → 风险列表。
依赖 week_stats 产出的 stats this_week / last_week以及 milestone_progress
milestones 列表commits 用于 bus_factor 复算可选避免 stats 内无明细
"""
now_utc = (now or datetime.now(UTC)).astimezone(UTC) \
if (now or datetime.now(UTC)).tzinfo else (now or datetime.now(UTC)).replace(tzinfo=UTC)
this_start = now_utc - timedelta(days=7)
warnings: list[dict] = []
tw: dict = stats.get("this_week", {})
# 1) 低活跃
commits_this = tw.get("commits", 0)
if commits_this < LOW_ACTIVITY_COMMITS:
warnings.append({
"level": "warning", "type": "low_activity",
"message": f"本周提交仅 {commits_this} 次(低于阈值 {LOW_ACTIVITY_COMMITS}),项目可能进展缓慢",
"metric": commits_this, "suggestion": "确认是否进入收尾阶段;若无,组织一次进度同步。",
})
# 2) bus_factor单一贡献者本周提交占比 > 50%
if commits:
w_commits = _commits_in(commits, this_start, now_utc)
else:
w_commits = [] # 无明细 → 无法判 bus factor
if w_commits:
login_counts: Counter = Counter(_author_login(it) for it in w_commits)
top_login, top_n = login_counts.most_common(1)[0]
ratio = top_n / len(w_commits)
active_n = len({lg for lg in login_counts if lg})
if ratio > BUS_FACTOR_RATIO and active_n <= 2:
warnings.append({
"level": "critical", "type": "bus_factor",
"message": f"bus factor 风险:{top_login or '(匿名)'} 一人贡献本周 {top_n}/{len(w_commits)} "
f"次提交({round(ratio*100)}%),活跃贡献者仅 {active_n}",
"metric": round(ratio, 3), "suggestion": "引入第二贡献者 / 文档化核心模块,降低单点依赖。",
})
# 3) stale issue / stale PR沿用 week_stats 已统计的口径)
stale_iss = tw.get("issues_stale", 0)
if stale_iss >= 5:
warnings.append({
"level": "warning" if stale_iss < 20 else "critical",
"type": "stale_issue",
"message": f"存在 {stale_iss} 个开放 Issue 超过 {STALE_ISSUE_DAYS} 天无活动",
"metric": stale_iss, "suggestion": "分诊:关闭无效 Issue、分配负责人或拆解。",
})
stale_pr = tw.get("prs_open_stale", 0)
if stale_pr >= 1:
warnings.append({
"level": "warning" if stale_pr < 3 else "critical",
"type": "stale_pr",
"message": f"存在 {stale_pr} 个开放 PR 超过 {STALE_PR_DAYS} 天未 review/合并",
"metric": stale_pr, "suggestion": "安排 review 或明确 reject避免 PR 堆积。",
})
# 4) 逾期里程碑
for ms in milestones:
if isinstance(ms, dict) and ms.get("overdue"):
warnings.append({
"level": "critical", "type": "overdue_milestone",
"message": f"里程碑「{ms.get('name', '?')}」已逾期"
+ (f"due {ms.get('due_date')}" if ms.get("due_date") else "")
+ f",完成率 {ms.get('completion_pct', 0)}%",
"metric": ms.get("completion_pct", 0),
"suggestion": "重新评估范围或顺延 deadline并同步干系人。",
})
# 排序critical > warning > info
rank = {"critical": 0, "warning": 1, "info": 2}
warnings.sort(key=lambda w: (rank.get(w["level"], 9), w["type"]))
return warnings
# ---------------------------------------------------------------------------
# 主入口(取数 + 算法)
# ---------------------------------------------------------------------------
def analyze(owner: str, repo: str, now: datetime | None = None) -> dict[str, Any]:
"""取数 + 算法:返回完整结果 dict。"""
if now is None:
now = datetime.now(UTC)
commits = c.commits(owner, repo, max_pages=10)
issues = c.issues_all(owner, repo)
prs = c.prs_all(owner, repo)
milestones = c.milestones(owner, repo)
contributors = c.contributors(owner, repo)
stats = week_stats(commits, issues, prs, contributors, now)
ms_progress = milestone_progress(milestones, issues, now)
warnings = risk_warnings(stats, ms_progress, contributors, commits=commits, now=now)
tr = trend(stats["this_week"], stats["last_week"])
return {
"scenario": "S5_progress_tracking",
"repo": f"{owner}/{repo}",
"generated_at": now.astimezone(UTC).isoformat(),
"week_stats": stats,
"trend": tr,
"milestones": ms_progress,
"risk_warnings": warnings,
"meta": {
"commits_fetched": len(commits),
"issues_fetched": len(issues),
"prs_fetched": len(prs),
"milestones_fetched": len(milestones),
"contributors_fetched": len(contributors),
},
}
# ---------------------------------------------------------------------------
# 渲染Markdown 周报
# ---------------------------------------------------------------------------
def render_report(result: dict[str, Any]) -> str:
repo = result["repo"]
stats = result["week_stats"]
tw, lw = stats["this_week"], stats["last_week"]
tr = result["trend"]
lines = [
f"# 科研进度智能跟踪周报 — {repo}\n",
f"> 场景 S5 · 子赛题四「应用 GitLink 辅助科研」· 生成于 {result['generated_at']}\n",
"## 一、本周 / 上周活动对比\n",
"| 指标 | 本周 | 上周 |",
"|------|------|------|",
f"| 提交 commits | {tw['commits']} | {lw['commits']} |",
f"| Issue 新增 | {tw['issues_opened']} | {lw['issues_opened']} |",
f"| Issue 关闭 | {tw['issues_closed']} | {lw['issues_closed']} |",
f"| 开放 stale issue (>{STALE_ISSUE_DAYS}天) | {tw['issues_stale']} | {lw['issues_stale']} |",
f"| PR 新增 | {tw['prs_opened']} | {lw['prs_opened']} |",
f"| PR 合并 | {tw['prs_merged']} | {lw['prs_merged']} |",
f"| 开放 stale PR (>{STALE_PR_DAYS}天) | {tw['prs_open_stale']} | {lw['prs_open_stale']} |",
f"| 活跃贡献者 | {tw['contributors_active']} | {lw['contributors_active']} |",
"",
f"- **趋势**commit 周环比 **{tr['commit_delta_pct']}%**,活跃度等级 `{tr['activity_level']}`",
"",
"## 二、里程碑进度\n",
]
ms = result["milestones"]
if ms:
lines += [
"| 里程碑 | 完成/总数 | 完成率 | due_date | 状态 |",
"|--------|-----------|--------|----------|------|",
]
for m in ms:
flag = " ⚠️逾期" if m["overdue"] else ""
lines.append(
f"| {m['name']}{flag} | {m['closed']}/{m['total']} | {m['completion_pct']}% "
f"| {m['due_date'] or ''} | {m['status'] or ''} |"
)
else:
lines.append("_仓库无里程碑数据_")
lines += ["\n## 三、风险预警\n"]
warns = result["risk_warnings"]
if warns:
lines += ["| 级别 | 类型 | 说明 | 建议 |", "|------|------|------|------|"]
for w in warns:
lines.append(f"| {w['level']} | {w['type']} | {w['message']} | {w['suggestion']} |")
else:
lines.append("_未触发风险阈值进度正常_")
lines += [
"\n## 四、附\n",
f"- 取数commits={result['meta']['commits_fetched']} "
f"issues={result['meta']['issues_fetched']} prs={result['meta']['prs_fetched']} "
f"milestones={result['meta']['milestones_fetched']} "
f"contributors={result['meta']['contributors_fetched']}",
f"- 窗口:本周 [{stats['window']['this_week_start']}, {stats['window']['now']}]"
f"上周 [{stats['window']['last_week_start']}, {stats['window']['last_week_end']})",
f"- 阈值stale_issue>{STALE_ISSUE_DAYS}天 / stale_pr>{STALE_PR_DAYS}天 / "
f"低活跃<{LOW_ACTIVITY_COMMITS}次/周 / bus_factor>{int(BUS_FACTOR_RATIO*100)}%",
"",
]
return "\n".join(lines)
# ---------------------------------------------------------------------------
def main():
ap = argparse.ArgumentParser(description="S5 科研进度智能跟踪与预警")
ap.add_argument("--owner", required=True)
ap.add_argument("--repo", required=True)
ap.add_argument("--out", help="输出目录(写 report.json + weekly_report.md省略则打印 JSON")
args = ap.parse_args()
result = analyze(args.owner, args.repo)
if args.out:
os.makedirs(args.out, exist_ok=True)
with open(os.path.join(args.out, "report.json"), "w", encoding="utf-8") as f:
json.dump(result, f, ensure_ascii=False, indent=2)
with open(os.path.join(args.out, "weekly_report.md"), "w", encoding="utf-8") as f:
f.write(render_report(result))
print(f"✓ S5 周报完成 → {args.out}/report.json | weekly_report.md")
print(f" 本周提交 {result['week_stats']['this_week']['commits']} "
f"(趋势 {result['trend']['activity_level']});风险 {len(result['risk_warnings'])}")
else:
print(json.dumps(result, ensure_ascii=False, indent=2))
if __name__ == "__main__":
main()

573
scripts/research/repro.py Normal file
View File

@ -0,0 +1,573 @@
"""repro.py — S3 科研项目合规与复现性检查。
输入一个科研仓库检查其合规性许可证版权依赖安全策略数据隐私
可复现性CI 配置lockfileREADME 是否含数据集/环境/构建说明版本 tag
密钥泄露分别给出 0-10 的复现分与合规分并产出检查清单风险项与中文报告
数据全部经 gitlink-cli 获取
- repo +info仓库信息版本 tag
- file +getLICENSE / README / go.mod / requirements.txt / package.json / .gitignore
- repo +tree data/.envconfig.gitea/.github workflows 等是否存在
用法
python repro.py --owner mindspore-Ecosystem --repo mindspore --out ./out
python repro.py --owner O --repo R # 仅打印 JSON
"""
from __future__ import annotations
import argparse
import json
import os
import re
import sys
from typing import Any
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import collect as c # noqa: E402
import gitlink_data as gd # noqa: E402
# ---------------------------------------------------------------------------
# 常量
# ---------------------------------------------------------------------------
# 复现性/合规相关的关键文件(相对仓库根)
KEY_FILES: tuple[str, ...] = (
"LICENSE", "LICENSE.txt", "LICENSE.md",
"README", "README.md", "README.rst",
"go.mod", "requirements.txt", "package.json", "Cargo.toml",
".gitignore", "SECURITY.md", "CONTRIBUTING.md",
)
# CI 配置目录/文件(复现性信号)。
# 注意:根 tree 通常只列顶层目录(.github/.devops/.gitea不一定展开到 workflows/
# 故同时收录顶层目录名与深层路径;.devops 是 GitLink 专属 CI/CD 目录。
CI_PATHS: tuple[str, ...] = (
".devops",
".gitea", ".gitea/workflows",
".github", ".github/workflows",
".gitlab-ci.yml", ".circleci", ".travis.yml", "azure-pipelines.yml",
)
# 锁文件(复现性信号:依赖版本固定)
LOCKFILES: tuple[str, ...] = (
"go.sum", "package-lock.json", "yarn.lock", "pnpm-lock.yaml",
"requirements.txt", "poetry.lock", "Pipfile.lock", "Cargo.lock", "composer.lock",
)
# 敏感目录/文件(数据隐私信号)
SENSITIVE_PATHS: tuple[str, ...] = (
"data/", "data", "dataset/", "datasets/", ".env", ".env.local",
"config/secrets", "secrets",
)
# README 复现性关键词(数据集 / 环境 / 构建 / 运行说明)
README_REPRO_KEYWORDS: tuple[str, ...] = (
"install", "setup", "环境", "依赖", "build", "构建", "运行", "run",
"dataset", "数据集", "docker", "conda", "pip install", "npm install",
"requirements", "reproduce", "复现", "环境配置", "usage", "用法",
)
# 许可证识别关键词(顺序即优先级)
LICENSE_PATTERNS: tuple[tuple[str, str], ...] = (
("MulanPSL", "MulanPSL-2.0"),
("木兰宽松许可证", "MulanPSL-2.0"),
("Apache License", "Apache-2.0"),
("MIT License", "MIT"),
("GNU GENERAL PUBLIC LICENSE", "GPL"),
("GNU Lesser General Public License", "LGPL"),
("BSD ", "BSD"),
("ISC License", "ISC"),
("Mozilla Public License", "MPL"),
("Unlicense", "Unlicense"),
)
# 密钥/敏感信息正则(按类别)
SECRET_PATTERNS: tuple[tuple[str, str, str], ...] = (
# (category, level, regex)
("private_key", "critical", r"-----BEGIN (?:RSA |EC |DSA |OPENSSH |PGP )?PRIVATE KEY-----"),
("aws_access_key", "critical", r"AKIA[0-9A-Z]{16}"),
("aws_secret", "critical", r"aws_secret_access_key\s*[:=]\s*['\"]?[A-Za-z0-9/+=]{40}"),
("generic_api_key", "high", r"(?i)api[_-]?key\s*[:=]\s*['\"]?[A-Za-z0-9_\-]{16,}"),
("google_api_key", "high", r"AIza[0-9A-Za-z_\-]{35}"),
("slack_token", "high", r"xox[baprs]-[0-9A-Za-z-]{10,}"),
("github_token", "high", r"gh[pousr]_[A-Za-z0-9]{36,}"),
("jwt", "medium", r"eyJ[A-Za-z0-9_\-]+\.eyJ[A-Za-z0-9_\-]+\.[A-Za-z0-9_\-]+"),
("email", "low", r"[a-zA-Z0-9._%+\-]+@[a-zA-Z0-9.\-]+\.[A-Za-z]{2,}"),
# 中国大陆手机号
("phone_cn", "low", r"(?<!\d)1[3-9]\d{9}(?!\d)"),
)
# ---------------------------------------------------------------------------
# 算法:纯函数(单测对象,不联网)
# ---------------------------------------------------------------------------
def identify_license(text: str) -> dict[str, Any]:
"""关键词匹配 LICENSE 文本,返回许可证信息。
返回: {"license": str, "recognized": bool, "evidence": str}
未识别返回 license="None"
"""
if not text or not text.strip():
return {"license": "None", "recognized": False,
"evidence": "LICENSE 文件为空或缺失"}
low = text.lower()
for pat, name in LICENSE_PATTERNS:
if pat.lower() in low:
# 找到关键词所在行作为证据
idx = low.find(pat.lower())
line_start = text.rfind("\n", 0, idx) + 1
line_end = text.find("\n", idx)
if line_end == -1:
line_end = len(text)
evidence = text[line_start:line_end].strip()[:120]
return {"license": name, "recognized": True, "evidence": evidence}
return {"license": "None", "recognized": False,
"evidence": "未匹配到已知许可证关键词"}
def scan_secrets(text: str, file: str = "") -> list[dict[str, Any]]:
"""扫描文本中的密钥/敏感信息。
返回: [{"level","category","file","line","detail"}, ...]
"""
if not text:
return []
findings: list[dict[str, Any]] = []
lines = text.splitlines()
for category, level, pattern in SECRET_PATTERNS:
for m in re.finditer(pattern, text):
# 计算行号与所在行内容
line_no = text.count("\n", 0, m.start()) + 1
line_content = lines[line_no - 1] if 0 <= line_no - 1 < len(lines) else ""
detail = m.group(0)
# 脱敏:长串截断
if len(detail) > 40:
detail = detail[:20] + "..." + detail[-6:]
findings.append({
"level": level, "category": category, "file": file,
"line": line_no, "detail": detail,
"context": line_content.strip()[:80],
})
return findings
def _tree_paths(tree: list) -> list[str]:
"""从 tree 列表提取所有路径字符串(兼容多种字段名)。"""
paths: list[str] = []
for item in tree:
if isinstance(item, dict):
for k in ("path", "name", "filepath"):
v = item.get(k)
if isinstance(v, str) and v:
paths.append(v)
break
elif isinstance(item, str) and item:
paths.append(item)
return paths
def _has_path(paths: list[str], targets: tuple[str, ...]) -> list[str]:
"""paths 中命中任一 target前缀/精确都算)的命中项,返回命中原文(去重保序)。"""
hits: list[str] = []
seen: set[str] = set()
low_targets = [t.lower().rstrip("/") for t in targets]
for p in paths:
pl = p.lower().rstrip("/")
for tl in low_targets:
if pl == tl or pl.startswith(tl + "/"):
key = f"{p}::{tl}"
if key not in seen:
seen.add(key)
hits.append(p)
break
return hits
def repro_checks(file_texts: dict[str, str], tree: list,
repo_info: dict | None = None) -> list[dict[str, Any]]:
"""复现性检查。
file_texts: {路径: 文本内容}已取好
tree: tree 列表已取好
repo_info: 仓库信息取版本 tag tag 字段则 unknown
返回检查项列表: [{"name","pass","score","evidence"}]
每项 score 0-20=缺失/失败, 1=部分, 2=完备
"""
paths = _tree_paths(tree)
items: list[dict[str, Any]] = []
# 1) CI 配置
ci_hits = _has_path(paths, CI_PATHS)
if ci_hits:
items.append({"name": "CI 配置", "pass": True, "score": 2,
"evidence": f"检测到 CI 配置: {', '.join(ci_hits[:3])}"})
else:
items.append({"name": "CI 配置", "pass": False, "score": 0,
"evidence": "未找到 .gitea/.github/.gitlab 等 CI 配置"})
# 2) lockfile依赖版本固定
lock_hits = _has_path(paths, LOCKFILES)
if lock_hits:
items.append({"name": "依赖锁文件", "pass": True, "score": 2,
"evidence": f"存在 lockfile: {', '.join(lock_hits[:3])}"})
else:
items.append({"name": "依赖锁文件", "pass": False, "score": 0,
"evidence": "未找到 go.sum/package-lock.json/poetry.lock 等锁文件"})
# 3) README 含数据集/环境/构建说明
readme_text = ""
for k in ("README.md", "README", "README.rst"):
if k in file_texts and file_texts[k]:
readme_text = file_texts[k]
break
if readme_text:
low = readme_text.lower()
hit_kw = [kw for kw in README_REPRO_KEYWORDS if kw.lower() in low]
score = 2 if len(hit_kw) >= 4 else (1 if len(hit_kw) >= 1 else 0)
items.append({"name": "README 复现说明", "pass": score > 0, "score": score,
"evidence": (f"README 含复现关键词 {len(hit_kw)} 个: {', '.join(hit_kw[:5])}"
if hit_kw else "README 存在但缺少数据集/环境/构建说明")})
else:
items.append({"name": "README 复现说明", "pass": False, "score": 0,
"evidence": "未找到 README"})
# 4) 版本 tag用 repo_info无 tag 字段则 unknown
info = repo_info or {}
tag = (info.get("version") or info.get("tag") or info.get("release_tag")
or info.get("default_branch") or "")
# 多数 GitLink repo_info 无显式 tag 字段 → 标 unknown不扣分但提示
has_tag = bool(info.get("version") or info.get("tag") or info.get("release_tag"))
if has_tag:
items.append({"name": "版本 tag", "pass": True, "score": 2,
"evidence": f"版本/release tag: {tag}"})
else:
items.append({"name": "版本 tag", "pass": False, "score": 1,
"evidence": f"repo_info 无显式 tag 字段(默认分支: {info.get('default_branch', 'unknown')}),建议打 tag 固定可复现版本"})
# 5) 容器化Dockerfile / docker-compose—— 复现环境
container_hits = _has_path(paths, ("Dockerfile", "docker-compose.yml",
"docker-compose.yaml", ".devcontainer"))
if container_hits:
items.append({"name": "容器化环境", "pass": True, "score": 2,
"evidence": f"存在容器配置: {', '.join(container_hits[:3])}"})
else:
items.append({"name": "容器化环境", "pass": False, "score": 0,
"evidence": "未找到 Dockerfile/docker-compose复现环境依赖手工描述"})
return items
def compliance_items(license_info: dict[str, Any], file_texts: dict[str, str],
tree: list) -> list[dict[str, Any]]:
"""合规性检查。
返回检查项列表: [{"name","pass","score","evidence"}]score 0-2
"""
paths = _tree_paths(tree)
items: list[dict[str, Any]] = []
# 1) LICENSE 声明
lic = license_info.get("license", "None")
recognized = license_info.get("recognized", False)
if recognized and lic != "None":
items.append({"name": "LICENSE 文件", "pass": True, "score": 2,
"evidence": f"LICENSE 声明为 {lic}"})
elif lic == "None" and not (file_texts.get("LICENSE") or file_texts.get("LICENSE.txt")
or file_texts.get("LICENSE.md")):
items.append({"name": "LICENSE 文件", "pass": False, "score": 0,
"evidence": "缺少 LICENSE 文件"})
else:
items.append({"name": "LICENSE 文件", "pass": False, "score": 1,
"evidence": "LICENSE 文件存在但类型未识别"})
# 2) SECURITY.md
sec_hits = _has_path(paths, ("SECURITY.md", "SECURITY", "security.md"))
if sec_hits:
items.append({"name": "安全策略 SECURITY.md", "pass": True, "score": 2,
"evidence": f"存在 {sec_hits[0]}"})
else:
items.append({"name": "安全策略 SECURITY.md", "pass": False, "score": 0,
"evidence": "缺少 SECURITY.md无安全披露流程"})
# 3) 版权头(采样 README/LICENSE 头部判断有无 Copyright
sample = (file_texts.get("LICENSE", "") + "\n" + file_texts.get("README.md", "")
+ "\n" + file_texts.get("README", ""))
has_copyright = ("copyright" in sample.lower()) or ("版权" in sample) or ("©" in sample)
if has_copyright:
items.append({"name": "版权声明", "pass": True, "score": 2,
"evidence": "LICENSE/README 中含 copyright/版权 声明"})
else:
items.append({"name": "版权声明", "pass": False, "score": 1,
"evidence": "未在 LICENSE/README 中发现版权声明(建议源文件头补 Copyright 注释)"})
# 4) 依赖合规(存在依赖清单即视为已声明,识别许可证更佳)
dep_present = bool(_has_path(paths, ("go.mod", "requirements.txt", "package.json",
"Cargo.toml", "pom.xml", "setup.py", "pyproject.toml")))
if dep_present:
items.append({"name": "依赖清单声明", "pass": True, "score": 2,
"evidence": "存在依赖管理文件(建议核对各依赖许可证兼容性)"})
else:
items.append({"name": "依赖清单声明", "pass": False, "score": 1,
"evidence": "未发现标准依赖管理文件"})
# 5) CONTRIBUTING.md社区合规
contrib_hits = _has_path(paths, ("CONTRIBUTING.md", "CONTRIBUTING", "contributing.md"))
if contrib_hits:
items.append({"name": "贡献指南", "pass": True, "score": 2,
"evidence": f"存在 {contrib_hits[0]}"})
else:
items.append({"name": "贡献指南", "pass": False, "score": 1,
"evidence": "缺少 CONTRIBUTING.md"})
return items
def data_privacy(tree: list, gitignore_text: str) -> dict[str, Any]:
"""数据隐私检查。
返回: {
"items": [{"name","pass","score","evidence"}],
"risks": [...], # 高风险项明细
}
"""
paths = _tree_paths(tree)
items: list[dict[str, Any]] = []
risks: list[str] = []
# 1) data/ 目录是否入库
data_hits = _has_path(paths, ("data/", "dataset/", "datasets/"))
if data_hits:
items.append({"name": "数据目录入库", "pass": False, "score": 0,
"evidence": f"data/ 目录已入库: {', '.join(data_hits[:3])}(建议大文件走外部存储/DVC"})
risks.append(f"数据目录入库: {', '.join(data_hits[:3])}(可能含敏感数据)")
else:
items.append({"name": "数据目录入库", "pass": True, "score": 2,
"evidence": "未发现 data/ 目录入库"})
# 2) .env 是否入库
env_hits = _has_path(paths, (".env", ".env.local", ".env.production"))
if env_hits:
items.append({"name": ".env 入库", "pass": False, "score": 0,
"evidence": f".env 已入库: {', '.join(env_hits[:3])}(高风险,疑似凭据泄露)"})
risks.append(f".env 已入库: {', '.join(env_hits[:3])}(凭据泄露风险)")
else:
items.append({"name": ".env 入库", "pass": True, "score": 2,
"evidence": ".env 未入库"})
# 3) .gitignore 是否忽略 .env
gi = (gitignore_text or "").lower()
ignores_env = ".env" in gi
if ignores_env:
items.append({"name": ".gitignore 忽略 .env", "pass": True, "score": 2,
"evidence": ".gitignore 已配置忽略 .env"})
else:
items.append({"name": ".gitignore 忽略 .env", "pass": False, "score": 1,
"evidence": ".gitignore 未忽略 .env建议添加 .env"})
if not env_hits:
risks.append(".gitignore 未忽略 .env预防性建议")
return {"items": items, "risks": risks}
def _score_10(items: list[dict[str, Any]], cap: float = 10.0) -> float:
"""把检查项的 0-2 分聚合为 0-10 分sum(score)/sum(max=2) * 10。"""
total = sum(it.get("score", 0) for it in items)
max_total = sum(2 for _ in items)
if max_total == 0:
return 0.0
return round(min(cap, total / max_total * cap), 1)
# ---------------------------------------------------------------------------
# 数据采集(调 gitlink-cli算法不依赖本节
# ---------------------------------------------------------------------------
def collect_file_texts(owner: str, repo: str, ref: str = "master") -> dict[str, str]:
"""批量取关键文件文本。缺失文件返回空串(不出现在 dict 中)。"""
out: dict[str, str] = {}
for path in KEY_FILES:
try:
txt = c.file_text(owner, repo, path, ref=ref)
except Exception:
txt = ""
if txt and txt.strip():
out[path] = txt
return out
def collect_tree(owner: str, repo: str, ref: str = "master") -> list:
"""取根 tree含扫 data/、.env、config、workflows 等)。"""
try:
return c.tree(owner, repo, path="", ref=ref)
except Exception:
return []
# ---------------------------------------------------------------------------
# 主流程
# ---------------------------------------------------------------------------
def run(owner: str, repo: str) -> dict[str, Any]:
info = c.repo_info(owner, repo)
ref = info.get("default_branch") or "master"
file_texts = collect_file_texts(owner, repo, ref)
tree = collect_tree(owner, repo, ref)
license_text = (file_texts.get("LICENSE") or file_texts.get("LICENSE.txt")
or file_texts.get("LICENSE.md") or "")
license_info = identify_license(license_text)
repro = repro_checks(file_texts, tree, repo_info=info)
compliance = compliance_items(license_info, file_texts, tree)
gitignore_text = file_texts.get(".gitignore", "")
dp = data_privacy(tree, gitignore_text)
# 汇总密钥扫描(扫所有已取文本 + gitignore
all_secrets: list[dict[str, Any]] = []
for path, txt in file_texts.items():
all_secrets.extend(scan_secrets(txt, file=path))
repro_score = _score_10(repro)
compliance_score = _score_10(compliance + dp["items"])
# 风险项汇总
risks: list[dict[str, Any]] = []
for it in repro + compliance + dp["items"]:
if not it["pass"]:
risks.append({"area": "repro/compliance", "name": it["name"],
"evidence": it["evidence"], "level": "medium"})
for r in dp["risks"]:
risks.append({"area": "privacy", "name": "数据隐私", "evidence": r, "level": "high"})
for s in all_secrets:
risks.append({"area": "secret", "name": s["category"], "file": s["file"],
"line": s["line"], "detail": s["detail"],
"level": s["level"]})
# 按级别排序
level_rank = {"critical": 0, "high": 1, "medium": 2, "low": 3}
risks.sort(key=lambda r: level_rank.get(r.get("level", "low"), 9))
return {
"scenario": "S3_compliance_reproducibility",
"repo": f"{owner}/{repo}",
"default_branch": ref,
"license": license_info["license"],
"repro_items": repro,
"compliance_items": compliance,
"privacy_items": dp["items"],
"secrets": all_secrets,
"risks": risks,
"repro_score": repro_score,
"compliance_score": compliance_score,
"meta": {"key_files_found": sorted(file_texts.keys()),
"tree_size": len(tree),
"languages": c.languages(owner, repo)},
}
# ---------------------------------------------------------------------------
# 渲染Markdown 报告
# ---------------------------------------------------------------------------
def _grade(score: float) -> str:
if score >= 8:
return "良好"
if score >= 6:
return "及格"
if score >= 4:
return "偏弱"
return "较差"
def render_report(result: dict[str, Any]) -> str:
repo = result["repo"]
lic = result["license"]
rs = result["repro_score"]
cs = result["compliance_score"]
lines = [
f"# 科研项目合规与复现性检查报告 — {repo}\n",
f"> 场景 S3 · 子赛题四「应用 GitLink 辅助科研」\n",
f"- **默认分支**: `{result.get('default_branch', 'master')}`",
f"- **识别许可证**: `{lic}`",
f"- **复现性评分**: **{rs}/10**{_grade(rs)}",
f"- **合规性评分**: **{cs}/10**{_grade(cs)}\n",
]
# 检查清单表
lines += ["## 一、复现性检查清单\n",
"| 检查项 | 通过 | 得分 | 证据 |",
"|--------|:----:|:----:|------|"]
for it in result["repro_items"]:
mark = "PASS" if it["pass"] else "FAIL"
lines.append(f"| {it['name']} | {mark} | {it['score']}/2 | {it['evidence']} |")
lines += ["\n## 二、合规性检查清单\n",
"| 检查项 | 通过 | 得分 | 证据 |",
"|--------|:----:|:----:|------|"]
for it in result["compliance_items"]:
mark = "PASS" if it["pass"] else "FAIL"
lines.append(f"| {it['name']} | {mark} | {it['score']}/2 | {it['evidence']} |")
lines += ["\n## 三、数据隐私检查\n",
"| 检查项 | 通过 | 得分 | 证据 |",
"|--------|:----:|:----:|------|"]
for it in result["privacy_items"]:
mark = "PASS" if it["pass"] else "FAIL"
lines.append(f"| {it['name']} | {mark} | {it['score']}/2 | {it['evidence']} |")
# 风险项表
risks = result["risks"]
lines += ["\n## 四、风险项(按严重程度排序)\n",
"| 级别 | 类别 | 名称 | 文件:行 | 证据 |",
"|:----:|------|------|---------|------|"]
if risks:
for r in risks:
lvl = r.get("level", "medium")
area = r.get("area", "")
name = r.get("name", "")
file_loc = ""
if r.get("file"):
file_loc = f"{r['file']}:{r.get('line', '')}"
ev = r.get("evidence") or r.get("detail", "")
lines.append(f"| {lvl} | {area} | {name} | {file_loc} | {ev} |")
else:
lines.append("| — | — | 无风险项 | — | 全部检查通过 |")
# 密钥小结
secrets = result.get("secrets", [])
if secrets:
lines.append(f"\n> 检出 **{len(secrets)}** 处疑似敏感信息(见风险项表),请人工复核确认。\n")
lines.append(f"\n_复现分 {rs}/10 · 合规分 {cs}/10 · 树节点 {result.get('meta',{}).get('tree_size','?')}_\n")
return "\n".join(lines)
# ---------------------------------------------------------------------------
def main():
ap = argparse.ArgumentParser(description="S3 科研项目合规与复现性检查")
ap.add_argument("--owner", required=True)
ap.add_argument("--repo", required=True)
ap.add_argument("--out", help="输出目录(写 repro.json + compliance_report.md省略则打印 JSON")
args = ap.parse_args()
result = run(args.owner, args.repo)
if args.out:
os.makedirs(args.out, exist_ok=True)
with open(os.path.join(args.out, "repro.json"), "w", encoding="utf-8") as f:
json.dump(result, f, ensure_ascii=False, indent=2)
with open(os.path.join(args.out, "compliance_report.md"), "w", encoding="utf-8") as f:
f.write(render_report(result))
print(f"✓ S3 合规/复现检查完成 → {args.out}/repro.json | compliance_report.md")
print(f" 许可证: {result['license']}")
print(f" 复现分: {result['repro_score']}/10 合规分: {result['compliance_score']}/10")
print(f" 风险项: {len(result['risks'])}")
else:
print(json.dumps(result, ensure_ascii=False, indent=2))
if __name__ == "__main__":
main()

View File

@ -0,0 +1,6 @@
# 子赛题四·科研辅助算法层依赖
# 数据采集/报告/匹配/进度/合规/洞悉 仅用 Python 标准库subprocess/json/sqlite3/re/datetime无需第三方库。
# 实测真正被 import 的第三方库只有以下两个:
networkx>=3.1 # S2 科研知识图谱构建
plotly>=5.18 # S6 科研成果可视化(交互 HTML
# (如后续 S4 升级为 TF-IDF、S6 改用模板渲染,再按需加 scikit-learn / jinja2

View File

@ -0,0 +1,272 @@
"""test_graph_build.py — S2 知识图谱的纯单元测试(不联网、不调 gitlink-cli
取数算法分离build_graph() 只吃已构造好的 mock 数据结构
运行`python test_graph_build.py` `pytest scripts/research/`
"""
import os
import sys
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import graph_build as G # noqa: E402
# ---------------------------------------------------------------------------
# mock 数据构造
# ---------------------------------------------------------------------------
def _mock_repos():
return [
{
"fullname": "alice/mindspore-vision",
"identifier": "mindspore-vision",
"author": {"login": "alice"},
"description": "deep learning library for computer vision, "
"image classification and object detection using CNN",
"language": {"name": "Python"},
"praises_count": 120,
"forked_count": 30,
},
{
"fullname": "bob/mindnlp",
"identifier": "mindnlp",
"author": {"login": "bob"},
"description": "A nlp library: transformer, bert, text classification",
"language": {"name": "Python"},
"praises_count": 200,
"forked_count": 50,
},
]
def _mock_contributors_map():
# alice 既贡献自己的 repo也贡献 bob 的 repo→ 核心学者 + collaborates_with
return {
"alice/mindspore-vision": [
{"login": "alice", "contribution_perc": "60.0%"},
{"login": "carol", "contribution_perc": "40.0%"},
{"login": "i-robot", "contribution_perc": "0.0%"}, # bot应被过滤
],
"bob/mindnlp": [
{"login": "alice", "contribution_perc": "10.0%"},
{"login": "bob", "contribution_perc": "90.0%"},
{"login": "dependabot", "contribution_perc": "1.0%"}, # bot
],
}
def _mock_readmes():
return {
"alice/mindspore-vision": "pytorch and resnet for object detection",
"bob/mindnlp": "pretrain gpt and llm models",
}
def _build():
return G.build_graph(_mock_repos(), _mock_contributors_map(),
{}, _mock_readmes(), keywords=["vision", "nlp"])
# ---------------------------------------------------------------------------
# 工具函数
# ---------------------------------------------------------------------------
def test_is_bot():
assert G.is_bot("i-robot") is True
assert G.is_bot("dependabot[bot]") is True
assert G.is_bot("alice") is False
assert G.is_bot("") is True
def test_parse_ratio():
assert G.parse_ratio("60.0%") == 0.6
assert abs(G.parse_ratio("0.5") - 0.5) < 1e-9
assert G.parse_ratio(0.3) == 0.3
assert G.parse_ratio(None) == 0.0
assert G.parse_ratio("n/a") == 0.0
# ---------------------------------------------------------------------------
# 建图:节点计数
# ---------------------------------------------------------------------------
def test_node_counts():
result = _build()
by_type = {}
for n in result["nodes"]:
by_type[n["type"]] = by_type.get(n["type"], 0) + 1
assert by_type.get("repo", 0) == 2
# alice / carol / bob = 3 个学者bot 被过滤)
assert by_type.get("scholar", 0) == 3
# 主题CV + deep_learning + nlp = 至少 3 个
assert by_type.get("topic", 0) >= 3
assert result["meta"]["repo_count"] == 2
assert result["meta"]["scholar_count"] == 3
def test_repo_node_props():
result = _build()
repo_nodes = [n for n in result["nodes"] if n["type"] == "repo"]
labels = {n["label"] for n in repo_nodes}
assert "alice/mindspore-vision" in labels
r = [n for n in repo_nodes if n["label"] == "alice/mindspore-vision"][0]
assert r["props"]["language"] == "Python"
assert r["props"]["stars"] == 120
assert r["props"]["forks"] == 30
# ---------------------------------------------------------------------------
# 建图:边
# ---------------------------------------------------------------------------
def test_contributes_to_weight():
result = _build()
edges = result["edges"]
contrib = [e for e in edges if e["type"] == "contributes_to"
and e["source"] == "scholar:alice"
and e["target"] == "repo:alice/mindspore-vision"]
assert contrib, "应有 alice→自己repo 的 contributes_to 边"
# 60% → 0.6
assert abs(contrib[0]["weight"] - 0.6) < 1e-9
def test_owns_edge():
result = _build()
owns = [e for e in result["edges"] if e["type"] == "owns"
and e["source"] == "scholar:bob"
and e["target"] == "repo:bob/mindnlp"]
assert owns, "bob 应有 owns 边到自己的 repo"
def test_covers_topic_weight_range():
"""covers_topic 权重应在 (0, 1] 且 <= 1。"""
result = _build()
covers = [e for e in result["edges"] if e["type"] == "covers_topic"]
assert covers, "应至少有一条 covers_topic 边"
for e in covers:
assert 0.0 <= e["weight"] <= 1.0
# 命中 computer_vision 主题description 含 object detection/image classification
cv_edges = [e for e in covers if e["target"] == "topic:computer_vision"]
assert cv_edges, "应识别出 computer_vision 主题"
def test_collaborates_with():
"""alice 与 carol 同在 alice/mindspore-vision → 至少一条 collaborates_with。"""
result = _build()
collab = [e for e in result["edges"] if e["type"] == "collaborates_with"]
assert collab, "应有 collaborates_with 边"
pair = {e["source"] for e in collab}
assert "scholar:alice" in pair
def test_related_to_when_topic_cooccur():
"""computer_vision 与 deep_learning 在同一 repo 共现 → related_to。"""
result = _build()
related = [e for e in result["edges"] if e["type"] == "related_to"
and e["source"] == "topic:computer_vision"]
# mindspore-vision 的 description 同时命中 CV + deep_learning
assert related, "共现主题应有 related_to 边"
assert any(e["target"] == "topic:deep_learning" for e in related)
# ---------------------------------------------------------------------------
# 衍生统计
# ---------------------------------------------------------------------------
def test_core_scholars():
result = _build()
# alice 出现在 2 个 repo → 排首位
top = result["core_scholars"][0]
assert top["login"] == "alice"
assert top["repo_count"] == 2
def test_topic_heat_top():
result = _build()
heat = result["topic_heat"]
assert heat, "应有主题热度榜"
# deep_learning 在两个 repo 都命中 → 应在前列
topics = [h["topic"] for h in heat]
assert "deep_learning" in topics
# ---------------------------------------------------------------------------
# 渲染
# ---------------------------------------------------------------------------
def test_render_mermaid_header():
result = _build()
mmd = G.render_mermaid(result)
assert "graph TD" in mmd
assert "classDef" in mmd # 着色定义
assert "```" in mmd
def test_render_mermaid_node_limit():
"""节点过多时应被截断到 node_limit。"""
big_repos = []
big_contribs = {}
big_readmes = {}
for i in range(60):
fn = f"u{i}/repo{i}"
big_repos.append({
"fullname": fn, "identifier": f"repo{i}",
"author": {"login": f"u{i}"}, "description": "deep learning",
"language": {"name": "Python"}, "praises_count": 0, "forked_count": 0,
})
big_contribs[fn] = [{"login": f"u{i}", "contribution_perc": "100%"}]
big_readmes[fn] = "deep learning"
result = G.build_graph(big_repos, big_contribs, {}, big_readmes,
keywords=["dl"])
mmd = G.render_mermaid(result, node_limit=40)
# mermaid 里出现的节点声明数应 <= 40
node_lines = [ln for ln in mmd.splitlines() if '["' in ln and "-->" not in ln]
assert len(node_lines) <= 40
def test_render_dot_header():
result = _build()
dot = G.render_dot(result)
assert dot.startswith("digraph G")
assert "fillcolor" in dot
def test_render_report_sections():
result = _build()
md = G.render_report(result)
assert "知识图谱报告" in md
assert "主题热度榜" in md
assert "核心学者" in md
assert "deep_learning" in md or "computer_vision" in md
def test_render_report_empty_safe():
result = {"scenario": "S2", "keywords": [], "nodes": [], "edges": [],
"core_scholars": [], "core_teams": [], "topic_heat": [],
"meta": {"repo_count": 0, "node_count": 0, "edge_count": 0,
"scholar_count": 0, "topic_count": 0}}
md = G.render_report(result)
assert "知识图谱报告" in md
assert "未识别" in md or "暂无" in md
def test_empty_inputs():
"""空输入不应抛异常。"""
result = G.build_graph([], {}, {}, {})
assert result["meta"]["node_count"] == 0
assert result["meta"]["edge_count"] == 0
assert result["nodes"] == []
# ---------------------------------------------------------------------------
def _run_all():
fns = [v for k, v in sorted(globals().items()) if k.startswith("test_")]
for fn in fns:
fn()
print(f"PASS {fn.__name__}")
print(f"\nAll {len(fns)} graph_build tests passed.")
if __name__ == "__main__":
_run_all()

View File

@ -0,0 +1,80 @@
"""test_helpers.py — gitlink_data / collect 归一化工具的纯单元测试(不联网)。
运行`pytest scripts/research/` `python scripts/research/test_helpers.py`
"""
import sys
import os
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import gitlink_data as gd # noqa: E402
import collect # noqa: E402
def test_first_list_list_input():
assert gd.first_list([1, 2, 3]) == [1, 2, 3]
def test_first_list_known_key():
assert gd.first_list({"projects": [{"id": 1}], "total_count": 1}) == [{"id": 1}]
def test_first_list_single_list_value():
# 无已知键但只有一个列表值时,兜底返回它
assert gd.first_list({"whatever": [9, 9]}) == [9, 9]
def test_first_list_empty():
assert gd.first_list({"total_count": 0}) == []
assert gd.first_list(None) == []
def test_total_count():
assert gd.total_count({"total_count": 42}) == 42
assert gd.total_count({"totalCount": 7}) == 7
assert gd.total_count({"projects": []}) is None
def test_login_of_flat():
assert collect.login_of({"login": "whale"}) == "whale"
def test_login_of_nested_author():
assert collect.login_of({"author": {"login": "baoerjun"}}) == "baoerjun"
def test_login_of_name_fallback():
assert collect.login_of({"name": "surponess"}) == "surponess"
def test_login_of_empty():
assert collect.login_of({}) == ""
assert collect.login_of("not a dict") == ""
def test_repo_fullname_with_author():
project = {"identifier": "gitlink-cli", "author": {"login": "whale_hihihi"}}
assert collect.repo_fullname(project) == "whale_hihihi/gitlink-cli"
def test_repo_fullname_missing_owner():
assert collect.repo_fullname({"identifier": "foo"}) == "foo"
def test_as_int_as_float():
assert collect.as_int("12") == 12
assert collect.as_int(None) == 0
assert collect.as_float("1.5") == 1.5
assert collect.as_float("x", -1.0) == -1.0
def _run_all():
fns = [v for k, v in sorted(globals().items()) if k.startswith("test_")]
for fn in fns:
fn()
print(f"PASS {fn.__name__}")
print(f"\nAll {len(fns)} helper tests passed.")
if __name__ == "__main__":
_run_all()

View File

@ -0,0 +1,260 @@
"""test_lineage.py — lineage.py 纯函数单元测试。
不联网不调 gitlink-cli取数算法分离算法函数接收已取好的
Python 数据结构本测试用 mock 数据喂算法
运行 python test_lineage.py
"""
from __future__ import annotations
import os
import sys
import unittest
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
# 仅 import 算法纯函数,绝不触发联网取数
import lineage as L # noqa: E402
class TestExperimentClassifier(unittest.TestCase):
def test_is_experiment_file_hits(self):
cases = [
"experiments/mnist/run.py",
"experiment_train.py",
"benchmark/imagenet/eval.py",
"eval/metrics.py",
"tests/test_model.py",
"test/foo.py",
"data/dataset.csv",
"src/benchmark_infer.py",
]
for p in cases:
with self.subTest(p=p):
self.assertTrue(L.is_experiment_file(p), f"应判为实验文件: {p}")
def test_is_experiment_file_misses(self):
cases = [
"src/model.py",
"README.md",
"docs/index.md",
"main.go",
"config.yaml",
"pkg/utils/util.py",
]
for p in cases:
with self.subTest(p=p):
self.assertFalse(L.is_experiment_file(p), f"不应判为实验文件: {p}")
def test_empty(self):
self.assertFalse(L.is_experiment_file(""))
self.assertFalse(L.is_experiment_file(None)) # type: ignore[arg-type]
class TestDocClassifier(unittest.TestCase):
def test_md_anywhere(self):
self.assertTrue(L.is_doc_file("README.md"))
self.assertTrue(L.is_doc_file("docs/guide.md"))
self.assertTrue(L.is_doc_file("deep/nested/notes.md"))
self.assertTrue(L.is_doc_file("GUIDE.MD")) # 大小写不敏感
def test_docs_dir(self):
self.assertTrue(L.is_doc_file("docs/index.html"))
self.assertTrue(L.is_doc_file("docs/config.yaml"))
self.assertTrue(L.is_doc_file("a/docs/b.txt"))
def test_non_doc(self):
self.assertFalse(L.is_doc_file("src/main.py"))
self.assertFalse(L.is_doc_file("tests/x.py"))
self.assertFalse(L.is_doc_file("data.csv"))
self.assertFalse(L.is_doc_file(""))
class TestTimestampParse(unittest.TestCase):
def test_iso_string(self):
t = L._to_epoch("2024-05-01T08:00:00Z")
self.assertGreater(t, 0)
# 反解回日期
self.assertTrue(L._iso_date(t).startswith("2024-05-01"))
def test_int_seconds(self):
self.assertAlmostEqual(L._to_epoch(1714521600), 1714521600.0)
def test_int_millis(self):
self.assertAlmostEqual(L._to_epoch(1714521600000), 1714521600.0)
def test_garbage(self):
self.assertEqual(L._to_epoch(None), 0.0)
self.assertEqual(L._to_epoch(""), 0.0)
self.assertEqual(L._to_epoch("not-a-date"), 0.0)
def test_iso_date_zero(self):
self.assertEqual(L._iso_date(0.0), "")
class TestBranchMap(unittest.TestCase):
def test_single_default_branch(self):
commits = [
{"timestamp": "2024-01-01T00:00:00Z"},
{"timestamp": "2024-06-01T00:00:00Z"},
{"timestamp": "2024-03-01T00:00:00Z"},
]
bm = L.build_branch_map(commits, "master")
self.assertEqual(len(bm), 1)
b = bm[0]
self.assertEqual(b["name"], "master")
self.assertTrue(b["is_default"])
self.assertEqual(b["commits"], 3)
# 最后活跃应取最大值 2024-06-01
self.assertTrue(b["last_active"].startswith("2024-06-01"))
def test_empty(self):
bm = L.build_branch_map([], "main")
self.assertEqual(bm, [{"name": "main", "commits": 0,
"last_active": "", "is_default": True}])
def test_non_list(self):
bm = L.build_branch_map(None, "main") # type: ignore[arg-type]
self.assertEqual(bm[0]["commits"], 0)
class TestCommitTimeline(unittest.TestCase):
def test_aggregation_and_sort(self):
commits = [
{"timestamp": "2024-01-02T00:00:00Z"},
{"timestamp": "2024-01-02T12:00:00Z"},
{"timestamp": "2024-01-01T00:00:00Z"},
]
tl = L.commit_timeline(commits, bucket="day")
self.assertEqual(tl, [
{"date": "2024-01-01", "count": 1},
{"date": "2024-01-02", "count": 2},
])
def test_skips_garbage(self):
commits = [{"timestamp": "bad"}, {"timestamp": ""}]
self.assertEqual(L.commit_timeline(commits), [])
class TestPrMergePatterns(unittest.TestCase):
def test_extract_and_sort(self):
prs = [
{"index": 10, "title": "feat A", "status": 1,
"pr_created_unix": 1717200000, "changed_files": 5},
{"index": 2, "title": "feat B", "status": 1,
"pr_merged_unix": 1714521600, "changed_files": 20},
{"index": 5, "title": "feat C", "status": 1,
"pr_created_unix": 1715000000, "additions": 3, "deletions": 4},
]
out = L.pr_merge_patterns(prs)
self.assertEqual(len(out), 3)
# 升序1714521600(2024-05-01) < 1715000000 < 1717200000
self.assertEqual(out[0]["number"], 2)
self.assertEqual(out[0]["changed_files"], 20)
self.assertTrue(out[0]["merged_time"].startswith("2024-05"))
# additions/deletions 兜底近似
last = next(p for p in out if p["number"] == 5)
self.assertEqual(last["changed_files"], 7)
# 空 merged_time 排末尾
no_time = L.pr_merge_patterns([
{"index": 1, "title": "x", "status": 1},
{"index": 2, "title": "y", "status": 1, "pr_created_unix": 1714521600},
])
self.assertEqual(no_time[-1]["number"], 1)
def test_empty(self):
self.assertEqual(L.pr_merge_patterns([]), [])
class TestDocEvolution(unittest.TestCase):
def test_filter_docs(self):
tree = [
{"name": "guide.md", "path": "docs/guide.md", "date": "2024-03-01"},
{"name": "index.html", "path": "docs/index.html"},
{"name": "model.py", "path": "src/model.py"}, # 排除
{"name": "old_2022-01-01.md", "path": "docs/old_2022-01-01.md"},
]
out = L.doc_evolution(tree)
files = {d["file"] for d in out}
self.assertIn("guide.md", files)
self.assertIn("index.html", files)
self.assertIn("old_2022-01-01.md", files)
self.assertNotIn("model.py", files)
# 文件名日期回退
old = next(d for d in out if d["file"] == "old_2022-01-01.md")
self.assertEqual(old["last_date"], "2022-01-01")
# 显式 date 字段优先
g = next(d for d in out if d["file"] == "guide.md")
self.assertEqual(g["last_date"], "2024-03-01")
class TestInnovationPoints(unittest.TestCase):
def test_high_impact_and_milestone(self):
prs = [
{"index": 1, "title": "chore: typo", "status": 1,
"pr_created_unix": 1714521600, "changed_files": 2}, # 低影响,无关键词
{"index": 2, "title": "feat: add transformer model", "status": 1,
"pr_created_unix": 1715000000, "changed_files": 25}, # 大规模+关键词
{"index": 3, "title": "implement benchmark suite", "status": 1,
"pr_created_unix": 1717200000, "changed_files": 4}, # 仅关键词
]
out = L.innovation_points(prs, commits=[])
# 应识别出 PR#2 和 PR#3PR#1 被过滤
nums = sorted(it["description"] for it in out)
self.assertTrue(any("transformer" in n.lower() for n in nums))
self.assertTrue(any("benchmark" in n.lower() for n in nums))
cats = [it["category"] for it in out]
self.assertIn("大规模重构/新特性", cats)
self.assertIn("特性引入", cats)
# 大规模 PR 排前impact 更高)
self.assertIn("transformer", out[0]["description"].lower())
# 每条都带证据
for it in out:
self.assertTrue(it["evidence"])
self.assertTrue(it["category"])
def test_empty(self):
self.assertEqual(L.innovation_points([], []), [])
def test_top_limit(self):
prs = [{"index": i, "title": f"add feature {i}", "status": 1,
"pr_created_unix": 1714521600 + i * 86400, "changed_files": 15}
for i in range(20)]
out = L.innovation_points(prs, commits=[], top=5)
self.assertEqual(len(out), 5)
class TestRender(unittest.TestCase):
"""渲染函数不抛异常、产出非空。"""
def _result(self):
return {
"scenario": "S1_repository_research_insight",
"repo": "o/r", "default_branch": "master",
"commit_timeline": [{"date": "2024-01-01", "count": 3}],
"branch_map": [{"name": "master", "commits": 5, "last_active": "2024-06-01",
"is_default": True}],
"pr_merge_patterns": [{"number": 2, "title": "feat A", "status": 1,
"merged_time": "2024-06-01", "changed_files": 9}],
"doc_evolution": [{"file": "guide.md", "last_date": "2024-03-01"}],
"experiment_files": ["benchmark/eval.py"],
"innovation_points": [{"description": "feat A", "evidence": "PR #2",
"category": "特性引入"}],
"meta": {"commit_count": 5, "merged_pr_count": 1, "doc_count": 1,
"experiment_file_count": 1},
}
def test_report(self):
r = L.render_report(self._result())
self.assertIn("仓库级科研项目洞悉报告", r)
self.assertIn("master", r)
self.assertIn("feat A", r)
def test_mermaid(self):
m = L.render_mermaid(self._result())
self.assertIn("gitGraph", m)
self.assertIn("master", m)
if __name__ == "__main__":
unittest.main(verbosity=2)

View File

@ -0,0 +1,77 @@
"""test_match.py — S4 协作匹配的纯单元测试(不联网)。
运行`pytest scripts/research/` `python scripts/research/test_match.py`
"""
import os
import sys
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import match as M # noqa: E402
def test_cosine_basic():
assert M.cosine({"a": 1, "b": 0}, {"a": 1, "b": 0}) == 1.0
assert M.cosine({"a": 1}, {"b": 1}) == 0.0
# 对称且在 [0,1]
v = M.cosine({"a": 1, "b": 2}, {"a": 2, "b": 1})
assert 0.0 < v < 1.0
def test_cosine_empty_safe():
assert M.cosine({}, {"a": 1}) == 0.0
assert M.cosine({}, {}) == 0.0
def test_jaccard():
assert M.jaccard(["python", "go"], ["python", "rust"]) == 1 / 3
assert M.jaccard([], ["x"]) == 0.0
def test_priority_weight():
assert M._priority_weight({"priority_name": ""}) == 3.0
assert M._priority_weight({"priority_name": "urgent"}) == 3.0
assert M._priority_weight({"priority_name": "普通"}) == 2.0
assert M._priority_weight({"priority_name": ""}) == 1.0
assert M._priority_weight({}) == 1.0
def test_parse_ratio_percent_string():
assert M._parse_ratio("1.18%") == 0.0118
assert abs(M._parse_ratio("50%") - 0.5) < 1e-9
def test_parse_ratio_plain():
assert M._parse_ratio(0.5) == 0.5
assert M._parse_ratio("0.2") == 0.2
assert M._parse_ratio(None) == 0.0
assert M._parse_ratio("n/a") == 0.0
def test_render_report_has_sections():
result = {
"repo": "o/r", "gap_topics": ["deep_learning"], "needed_languages": ["python"],
"gap_signals": [{"type": "unresolved_issue", "topic": "deep_learning",
"evidence": "x", "priority": ""}],
"candidates": [{"login": "alice", "score": 20.0, "topic_overlap": 0.5,
"language_match": 0.5, "activity_level": "high",
"repo_languages": ["python"], "reasons": ["覆盖缺口主题"]}],
"meta": {"pool_size": 1, "issue_sample": 1},
}
md = M.render_report(result)
assert "缺口分析" in md or "技术缺口" in md
assert "alice" in md
mm = M.render_mermaid(result)
assert mm.startswith("```mermaid") and "alice" in mm
def _run_all():
fns = [v for k, v in sorted(globals().items()) if k.startswith("test_")]
for fn in fns:
fn()
print(f"PASS {fn.__name__}")
print(f"\nAll {len(fns)} match tests passed.")
if __name__ == "__main__":
_run_all()

View File

@ -0,0 +1,150 @@
"""test_report.py — S5 进度跟踪与预警的纯单元测试(不联网)。
运行`pytest scripts/research/` `python scripts/research/test_report.py`
"""
import os
import sys
from datetime import datetime, timedelta, timezone
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import report as R # noqa: E402
UTC = timezone.utc
NOW = datetime(2026, 6, 29, 12, 0, 0, tzinfo=UTC)
def iso(dt):
return dt.isoformat()
def test_parse_time_iso():
assert R.parse_time("2026-06-01T08:30:00+00:00").year == 2026
assert R.parse_time("2026-06-01T08:30:00Z").month == 6
def test_parse_time_unix():
# 整数与整数字串都按秒级时间戳解析
dt = R.parse_time(1717200000)
assert dt is not None and dt.year == 2024
assert R.parse_time("1717200000").year == 2024
def test_parse_time_none_and_garbage():
assert R.parse_time(None) is None
assert R.parse_time("") is None
assert R.parse_time("not a date") is None
def test_in_window():
lo = NOW - timedelta(days=7)
assert R.in_window(NOW - timedelta(days=2), 7, NOW) is True
assert R.in_window(NOW - timedelta(days=10), 7, NOW) is False
assert R.in_window(None, 7, NOW) is False
def test_week_stats_splits_this_and_last_week():
commits = [
{"timestamp": iso(NOW - timedelta(days=2)), "author": {"login": "a"}}, # 本周
{"timestamp": iso(NOW - timedelta(days=3)), "author": {"login": "b"}}, # 本周
{"timestamp": iso(NOW - timedelta(days=10)), "author": {"login": "a"}}, # 上周
]
issues = [
{"created_at": iso(NOW - timedelta(days=1)), "status": "open"}, # 本周新增
{"created_at": iso(NOW - timedelta(days=9)), "status": "open"}, # 上周新增
]
prs = []
stats = R.week_stats(commits, issues, prs, ["a", "b"], NOW)
assert stats["this_week"]["commits"] == 2
assert stats["last_week"]["commits"] == 1
assert stats["this_week"]["issues_opened"] == 1
assert stats["last_week"]["issues_opened"] == 1
def test_week_stats_stale_issue():
# 开放、最近活动 > 30 天 → stale
old = NOW - timedelta(days=40)
issues = [{"status": "open", "created_at": iso(old), "journals_updated_at": iso(old)}]
stats = R.week_stats([], issues, [], [], NOW)
assert stats["this_week"]["issues_stale"] == 1
def test_trend():
assert R.trend({"commits": 10}, {"commits": 5})["activity_level"] == "increasing"
assert R.trend({"commits": 2}, {"commits": 10})["activity_level"] == "decreasing"
assert R.trend({"commits": 10}, {"commits": 10})["activity_level"] == "stable"
# 上周为 0、本周有提交 → 100%
assert R.trend({"commits": 3}, {"commits": 0})["commit_delta_pct"] == 100.0
def test_risk_low_activity():
stats = {"this_week": {"commits": 1, "issues_stale": 0, "prs_open_stale": 0}}
warns = R.risk_warnings(stats, [], [], commits=None, now=NOW)
assert any(w["type"] == "low_activity" for w in warns)
def test_risk_bus_factor():
# 一人占本周全部提交 → bus factor
commits = [{"timestamp": iso(NOW - timedelta(days=1)), "author": {"login": "only"}} for _ in range(5)]
stats = {"this_week": {"commits": 5, "issues_stale": 0, "prs_open_stale": 0}}
warns = R.risk_warnings(stats, [], [], commits=commits, now=NOW)
assert any(w["type"] == "bus_factor" for w in warns)
def test_milestone_progress_overdue():
ms = [{"name": "v1.0", "status": "open", "effective_date": iso(NOW - timedelta(days=5))}]
issues = [
{"milestone_name": "v1.0", "status": "closed"},
{"milestone_name": "v1.0", "status": "open"},
{"milestone_name": "v1.0", "status": "open"},
]
out = R.milestone_progress(ms, issues, NOW)
assert len(out) == 1
assert out[0]["total"] == 3 and out[0]["closed"] == 1
assert out[0]["completion_pct"] == round(100 / 3, 1)
assert out[0]["overdue"] is True
def test_render_report_contains_sections():
# 直接构造一个最小 result 喂渲染器(不触网)
res = {
"repo": "o/r", "generated_at": iso(NOW),
"week_stats": {"this_week": {"commits": 1, "issues_opened": 0, "issues_closed": 0,
"issues_stale": 0, "prs_opened": 0, "prs_merged": 0,
"prs_open_stale": 0, "contributors_active": 1},
"last_week": {"commits": 0, "issues_opened": 0, "issues_closed": 0,
"issues_stale": 0, "prs_opened": 0, "prs_merged": 0,
"prs_open_stale": 0, "contributors_active": 0},
"window": {"this_week_start": iso(NOW), "now": iso(NOW),
"last_week_start": iso(NOW), "last_week_end": iso(NOW)},
"total_contributors": 1},
"trend": {"commit_delta_pct": 100.0, "activity_level": "increasing"},
"milestones": [], "risk_warnings": [],
"meta": {"commits_fetched": 1, "issues_fetched": 0, "prs_fetched": 0,
"milestones_fetched": 0, "contributors_fetched": 1},
}
md = R.render_report(res)
assert "周报" in md and "趋势" in md
def test_pr_status_string_and_int():
# GitLink PR 列表 status 是字符串 'merged'/'open'/'closed'
assert R._pr_status({"status": "merged"}) == 1
assert R._pr_status({"status": "open"}) == 0
assert R._pr_status({"status": "closed"}) == 2
# 详情/health 可能给 pull_request_status 整数 0/1/2
assert R._pr_status({"pull_request_status": 1}) == 1
assert R._pr_status({"pull_request_status": 0}) == 0
assert R._pr_status({"pull_request_status": 2}) == 2
def _run_all():
fns = [v for k, v in sorted(globals().items()) if k.startswith("test_")]
for fn in fns:
fn()
print(f"PASS {fn.__name__}")
print(f"\nAll {len(fns)} report tests passed.")
if __name__ == "__main__":
_run_all()

View File

@ -0,0 +1,278 @@
"""test_repro.py — repro.py 的纯单元测试。
不联网不调 gitlink-cli"取数""算法"分离直接给算法函数喂 mock 数据
运行 python test_repro.py
"""
from __future__ import annotations
import os
import sys
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import repro # noqa: E402
# 用一个轻量测试框架,避免依赖 unittest保持无第三方依赖
_failures: list[str] = []
def check(cond: bool, msg: str) -> None:
status = "PASS" if cond else "FAIL"
print(f" [{status}] {msg}")
if not cond:
_failures.append(msg)
def check_eq(actual, expected, msg: str) -> None:
ok = actual == expected
status = "PASS" if ok else "FAIL"
print(f" [{status}] {msg} (got={actual!r})")
if not ok:
_failures.append(f"{msg}: expected {expected!r}, got {actual!r}")
# ---------------------------------------------------------------------------
# 1) identify_license
# ---------------------------------------------------------------------------
def test_identify_license():
print("\n== test_identify_license ==")
cases = {
"Apache License\nVersion 2.0": "Apache-2.0",
"MIT License\n\nCopyright (c) 2024": "MIT",
"GNU GENERAL PUBLIC LICENSE\nVersion 3": "GPL",
"木兰宽松许可证, 第2版": "MulanPSL-2.0",
"MulanPSL v2": "MulanPSL-2.0",
"BSD 3-Clause License": "BSD",
"ISC License": "ISC",
"Mozilla Public License Version 2.0": "MPL",
"random text without any license keyword": "None",
"": "None",
" \n ": "None",
}
for text, expected in cases.items():
info = repro.identify_license(text)
check_eq(info["license"], expected, f"identify_license({text[:24]!r})")
# 识别标志
check(repro.identify_license("MIT License")["recognized"] is True, "MIT recognized=True")
check(repro.identify_license("nope")["recognized"] is False, "unknown recognized=False")
# GPL 优先于 LGPLLGPL 文本应命中 LGPL因为 LGPL 模式在 GPL 之前)
l = repro.identify_license("GNU Lesser General Public License v3")
check_eq(l["license"], "LGPL", "LGPL not misidentified as GPL")
# ---------------------------------------------------------------------------
# 2) scan_secrets
# ---------------------------------------------------------------------------
def test_scan_secrets():
print("\n== test_scan_secrets ==")
sample = """\
API_KEY=sk_live_abcdef1234567890abcd
AWS_KEY=AKIAIOSFODNN7EXAMPLE
mail: someone@example.com
phone: 13812345678
-----BEGIN RSA PRIVATE KEY-----
MIIEpAIBAAKCAQEA...
"""
findings = repro.scan_secrets(sample, file="config.env")
cats = {f["category"] for f in findings}
check("generic_api_key" in cats, "detect generic_api_key")
check("aws_access_key" in cats, "detect aws_access_key")
check("email" in cats, "detect email")
check("phone_cn" in cats, "detect phone_cn")
check("private_key" in cats, "detect private_key")
# 每条都有必要字段
for f in findings:
check(set(["level", "category", "file", "line", "detail"]).issubset(f.keys()),
f"finding {f['category']} has required keys")
check_eq(f["file"], "config.env", f"{f['category']} file field")
# critical 级别存在
levels = {f["level"] for f in findings}
check("critical" in levels, "critical level present (private key/aws)")
# 空文本
check_eq(repro.scan_secrets(""), [], "empty text -> no findings")
# 无敏感信息文本
check_eq(repro.scan_secrets("hello world\nnothing here"), [], "clean text -> no findings")
# 长串脱敏detail 不超长)
long_token = "api_key=" + "a" * 60
f2 = repro.scan_secrets(long_token, file="x")
if f2:
check(len(f2[0]["detail"]) <= 50, "long secret is masked in detail")
# ---------------------------------------------------------------------------
# 3) repro_checks
# ---------------------------------------------------------------------------
def test_repro_checks():
print("\n== test_repro_checks ==")
file_texts = {
"README.md": "# Project\n\n## Install\n pip install -e .\n## Build\ndocker build .\n"
"Dataset: download from xxx. reproduce: python run.py\nUsage: see docs.",
}
tree = [
{"path": ".gitea/workflows/ci.yml"},
{"path": "go.sum"},
{"path": "requirements.txt"},
{"path": "Dockerfile"},
{"path": "src/main.go"},
]
items = repro.repro_checks(file_texts, tree, repo_info={"default_branch": "master"})
names = {it["name"]: it for it in items}
check(names["CI 配置"]["pass"] is True, "CI detected")
check_eq(names["CI 配置"]["score"], 2, "CI score=2")
check(names["依赖锁文件"]["pass"] is True, "lockfile detected")
check(names["README 复现说明"]["pass"] is True, "README repro keywords detected")
check(names["README 复现说明"]["score"] >= 1, "README repro score>=1")
check(names["容器化环境"]["pass"] is True, "Dockerfile detected")
# 无 CI / 无 lockfile 场景
items2 = repro.repro_checks({}, [], repo_info={})
names2 = {it["name"]: it for it in items2}
check(names2["CI 配置"]["pass"] is False, "no CI -> fail")
check_eq(names2["CI 配置"]["score"], 0, "no CI score=0")
check(names2["依赖锁文件"]["pass"] is False, "no lockfile -> fail")
check(names2["README 复现说明"]["pass"] is False, "no README -> fail")
# 版本 tag无 tag 字段 → score=1不扣满分但提示
check_eq(names2["版本 tag"]["score"], 1, "unknown tag -> score=1")
# 有 tag 字段 → score=2
items3 = repro.repro_checks({}, [], repo_info={"version": "v1.2.3"})
names3 = {it["name"]: it for it in items3}
check_eq(names3["版本 tag"]["score"], 2, "version present -> score=2")
check(names3["版本 tag"]["pass"] is True, "version present -> pass")
# score 范围合法
for it in items:
check(0 <= it["score"] <= 2, f"{it['name']} score in [0,2]")
# ---------------------------------------------------------------------------
# 4) compliance_items
# ---------------------------------------------------------------------------
def test_compliance_items():
print("\n== test_compliance_items ==")
license_info = {"license": "MIT", "recognized": True, "evidence": "MIT License"}
file_texts = {"LICENSE": "MIT License\nCopyright (c) 2024 Test", "README.md": "see LICENSE"}
tree = [
{"path": "SECURITY.md"},
{"path": "CONTRIBUTING.md"},
{"path": "requirements.txt"},
]
items = repro.compliance_items(license_info, file_texts, tree)
names = {it["name"]: it for it in items}
check(names["LICENSE 文件"]["pass"] is True, "LICENSE recognized")
check_eq(names["LICENSE 文件"]["score"], 2, "LICENSE score=2")
check(names["安全策略 SECURITY.md"]["pass"] is True, "SECURITY.md present")
check(names["版权声明"]["pass"] is True, "copyright present")
check(names["贡献指南"]["pass"] is True, "CONTRIBUTING present")
check(names["依赖清单声明"]["pass"] is True, "dep manifest present")
# 缺失场景
license_none = {"license": "None", "recognized": False, "evidence": "missing"}
items2 = repro.compliance_items(license_none, {}, [])
names2 = {it["name"]: it for it in items2}
check(names2["LICENSE 文件"]["pass"] is False, "no LICENSE -> fail")
check_eq(names2["LICENSE 文件"]["score"], 0, "no LICENSE score=0")
check(names2["安全策略 SECURITY.md"]["pass"] is False, "no SECURITY -> fail")
# ---------------------------------------------------------------------------
# 5) data_privacy
# ---------------------------------------------------------------------------
def test_data_privacy():
print("\n== test_data_privacy ==")
# 健康场景:无 data/,无 .env.gitignore 忽略 .env
tree = [{"path": "src/main.py"}, {"path": ".gitignore"}]
dp = repro.data_privacy(tree, ".env\n*.key\nnode_modules/")
items = {it["name"]: it for it in dp["items"]}
check(items["数据目录入库"]["pass"] is True, "no data dir -> pass")
check(items[".env 入库"]["pass"] is True, "no .env -> pass")
check(items[".gitignore 忽略 .env"]["pass"] is True, "gitignore ignores .env")
check_eq(len(dp["risks"]), 0, "healthy repo -> 0 risks")
# 风险场景data/ 入库,.env 入库gitignore 未忽略 .env
tree2 = [{"path": "data/raw.csv"}, {"path": ".env"}, {"path": "config/secrets.yml"}]
dp2 = repro.data_privacy(tree2, "node_modules/\n*.log")
items2 = {it["name"]: it for it in dp2["items"]}
check(items2["数据目录入库"]["pass"] is False, "data dir tracked -> fail")
check(items2[".env 入库"]["pass"] is False, ".env tracked -> fail")
check(items2[".gitignore 忽略 .env"]["pass"] is False, "gitignore missing .env")
check(len(dp2["risks"]) >= 1, "risky repo -> has risks")
# ---------------------------------------------------------------------------
# 6) 打分 _score_10
# ---------------------------------------------------------------------------
def test_score():
print("\n== test_score ==")
# 5 项全 2 分 → 10
full = [{"score": 2}] * 5
check_eq(repro._score_10(full), 10.0, "all pass -> 10")
# 5 项全 0 → 0
zero = [{"score": 0}] * 5
check_eq(repro._score_10(zero), 0.0, "all fail -> 0")
# 混合5 项中 3×2 + 2×0 = 6/10 = 6.0
mixed = [{"score": 2}, {"score": 2}, {"score": 2}, {"score": 0}, {"score": 0}]
check_eq(repro._score_10(mixed), 6.0, "mixed 6/10 -> 6.0")
# 空列表
check_eq(repro._score_10([]), 0.0, "empty -> 0")
# cap 不超 10
over = [{"score": 2}] * 8
check(repro._score_10(over) <= 10.0, "capped at 10")
# ---------------------------------------------------------------------------
# 7) render_report 端到端(用 mock 结果)
# ---------------------------------------------------------------------------
def test_render_report():
print("\n== test_render_report ==")
mock = {
"scenario": "S3_compliance_reproducibility",
"repo": "o/r",
"default_branch": "master",
"license": "MIT",
"repro_items": [{"name": "CI 配置", "pass": True, "score": 2, "evidence": "ci"}],
"compliance_items": [{"name": "LICENSE 文件", "pass": True, "score": 2, "evidence": "MIT"}],
"privacy_items": [{"name": ".env 入库", "pass": True, "score": 2, "evidence": "ok"}],
"secrets": [],
"risks": [],
"repro_score": 10.0,
"compliance_score": 10.0,
"meta": {"tree_size": 5},
}
md = repro.render_report(mock)
check("# 科研项目合规与复现性检查报告" in md, "report has title")
check("MIT" in md, "report shows license")
check("10.0/10" in md, "report shows scores")
check("复现性检查清单" in md, "report has repro checklist")
check("风险项" in md, "report has risk section")
# ---------------------------------------------------------------------------
def main():
test_identify_license()
test_scan_secrets()
test_repro_checks()
test_compliance_items()
test_data_privacy()
test_score()
test_render_report()
print()
if _failures:
print(f"RESULT: FAIL ({len(_failures)} failures)")
for m in _failures:
print(f" - {m}")
sys.exit(1)
else:
print("RESULT: ALL PASS")
if __name__ == "__main__":
main()

View File

@ -0,0 +1,406 @@
"""test_visual.py — S6 科研成果可视化的纯单元测试(不联网、不调 gitlink-cli
取数算法分离算法函数接收已构造好的 Python 数据结构mock commits/issues/...
测试只覆盖 bin_weekly / contribution_heatmap / extract_paper_links / classify_artifacts
不测 plotly 渲染
运行`python scripts/research/test_visual.py`
"""
import os
import sys
from datetime import datetime, timedelta, timezone
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import visual as V # noqa: E402
# ---------------------------------------------------------------------------
# 辅助:构造「最近 N 天」的时间戳/ISO 字符串
# ---------------------------------------------------------------------------
def _days_ago_ts(days: int) -> float:
return (datetime.now(timezone.utc) - timedelta(days=days)).timestamp()
def _days_ago_iso(days: int) -> str:
return (datetime.now(timezone.utc) - timedelta(days=days)).strftime("%Y-%m-%dT%H:%M:%SZ")
# ---------------------------------------------------------------------------
# _to_timestamp
# ---------------------------------------------------------------------------
def test_to_timestamp_int_seconds():
assert V._to_timestamp(0) == 0.0
assert V._to_timestamp(1719500000) == 1719500000.0
assert V._to_timestamp("1719500000") == 1719500000.0
def test_to_timestamp_millis():
# 13 位 → 视为毫秒
assert abs(V._to_timestamp(1719500000000) - 1719500000.0) < 1e-3
def test_to_timestamp_iso():
ts = V._to_timestamp("2024-06-15T12:00:00Z")
assert ts > 0
# 无效字符串 → 0
assert V._to_timestamp("not-a-date") == 0.0
assert V._to_timestamp(None) == 0.0
assert V._to_timestamp("") == 0.0
# ---------------------------------------------------------------------------
# bin_weekly
# ---------------------------------------------------------------------------
def test_bin_weekly_shape_and_sum():
weeks = 4
commits = [
{"timestamp": _days_ago_iso(2)}, # 本周
{"timestamp": _days_ago_iso(9)}, # 上周
{"timestamp": _days_ago_iso(9)},
{"timestamp": _days_ago_iso(20)}, # 3 周前
]
out = V.bin_weekly(commits, weeks, V._commit_time)
assert set(out.keys()) == {"labels", "counts"}
assert len(out["labels"]) == weeks
assert len(out["counts"]) == weeks
assert sum(out["counts"]) == len(commits)
# 标签是 ISO 周(形如 2026-Wxx
assert all("-W" in lab for lab in out["labels"])
def test_bin_weekly_drops_out_of_window_and_invalid():
weeks = 3
commits = [
{"timestamp": _days_ago_iso(1)}, # 窗口内
{"timestamp": _days_ago_iso(365)}, # 太早(窗口外)→ 丢弃
{"timestamp": "garbage"}, # 非法 → 丢弃
{}, # 无时间字段 → 丢弃
]
out = V.bin_weekly(commits, weeks, V._commit_time)
assert sum(out["counts"]) == 1
def test_bin_weekly_issues_uses_created_at():
weeks = 2
issues = [
{"created_at": _days_ago_iso(3)}, # 窗口内
{"created_at": _days_ago_iso(40)}, # 窗口外(>2 周)→ 丢弃
]
out = V.bin_weekly(issues, weeks, V._issue_time)
assert sum(out["counts"]) == 1 # 只有 3 天前那条进窗口
def test_bin_weekly_prs_uses_pr_created_unix():
weeks = 2
prs = [
{"pr_created_unix": int(_days_ago_ts(2))},
{"pr_created_unix": int(_days_ago_ts(50))},
]
out = V.bin_weekly(prs, weeks, V._pr_time)
assert sum(out["counts"]) == 1
def test_bin_weekly_empty():
out = V.bin_weekly([], 5, V._commit_time)
assert len(out["labels"]) == 5
assert out["counts"] == [0, 0, 0, 0, 0]
# ---------------------------------------------------------------------------
# contribution_heatmap
# ---------------------------------------------------------------------------
def test_heatmap_matrix_shape():
weeks = 3
contributors = [
{"login": "alice", "contributions": 100},
{"login": "bob", "contributions": 5},
]
commits = [
{"timestamp": _days_ago_iso(2), "author": {"login": "alice"}},
{"timestamp": _days_ago_iso(2), "author": {"login": "alice"}},
{"timestamp": _days_ago_iso(9), "author": {"login": "bob"}},
]
hm = V.contribution_heatmap(contributors, commits, weeks)
assert set(hm.keys()) == {"users", "weeks", "matrix"}
assert hm["users"][:2] == ["alice", "bob"]
assert len(hm["matrix"]) == len(hm["users"])
for row in hm["matrix"]:
assert len(row) == weeks
# alice 行总和 = 2
alice_row = hm["matrix"][hm["users"].index("alice")]
assert sum(alice_row) == 2
bob_row = hm["matrix"][hm["users"].index("bob")]
assert sum(bob_row) == 1
def test_heatmap_includes_commit_authors_not_in_contributors():
weeks = 2
contributors = [{"login": "alice", "contributions": 1}]
commits = [
{"timestamp": _days_ago_iso(2), "author": {"login": "alice"}},
{"timestamp": _days_ago_iso(3), "author": {"login": "carol"}}, # 不在 contributors
]
hm = V.contribution_heatmap(contributors, commits, weeks)
assert "carol" in hm["users"]
def test_heatmap_top_users_cap():
weeks = 2
contributors = [{"login": f"u{i}", "contributions": i} for i in range(20)]
commits = [{"timestamp": _days_ago_iso(1), "author": {"login": f"u{i}"}}
for i in range(20)]
hm = V.contribution_heatmap(contributors, commits, weeks, top_users=5)
assert len(hm["users"]) <= 5
assert len(hm["matrix"]) == len(hm["users"])
def test_heatmap_empty():
hm = V.contribution_heatmap([], [], 4)
assert hm["users"] == []
assert hm["matrix"] == []
assert len(hm["weeks"]) == 4
# ---------------------------------------------------------------------------
# extract_paper_links
# ---------------------------------------------------------------------------
def test_extract_arxiv_url():
text = "See https://arxiv.org/abs/2401.00012 for details."
links = V.extract_paper_links(text)
assert len(links) == 1
assert links[0]["type"] == "arxiv"
assert links[0]["target"] == "https://arxiv.org/abs/2401.00012"
# snippet 截取自原文上下文(窗口较窄,断言前缀即可)
assert links[0]["source_text_snippet"].startswith("See https://arxiv.org")
def test_extract_arxiv_pdf_url():
text = "paper: https://arxiv.org/pdf/2305.12345.pdf"
links = V.extract_paper_links(text)
assert len(links) == 1
# 归一成 abs 形式
assert links[0]["target"] == "https://arxiv.org/abs/2305.12345"
def test_extract_arxiv_bare():
text = "We use arXiv:2103.07018 in our method."
links = V.extract_paper_links(text)
assert any(l["target"] == "https://arxiv.org/abs/2103.07018" for l in links)
def test_extract_doi_url():
text = "Cited from https://doi.org/10.1000/182"
links = V.extract_paper_links(text)
assert len(links) == 1
assert links[0]["type"] == "doi"
assert links[0]["target"] == "https://doi.org/10.1000/182"
def test_extract_doi_bare():
text = "Reference 10.1109/5.771073 shows that."
links = V.extract_paper_links(text)
assert len(links) == 1
assert links[0]["target"] == "https://doi.org/10.1109/5.771073"
def test_extract_dedup_same_id():
text = ("arxiv 1 https://arxiv.org/abs/2401.00012 "
"and again https://arxiv.org/abs/2401.00012")
links = V.extract_paper_links(text)
assert len(links) == 1
def test_extract_dedup_across_doi_forms():
# doi.org 形式与裸 DOI 视为同一条
text = "https://doi.org/10.1000/182 and bare 10.1000/182 again"
links = V.extract_paper_links(text)
# 同一 DOI 只出现一次
targets = [l["target"] for l in links]
assert targets.count("https://doi.org/10.1000/182") == 1
def test_extract_multiple_and_order():
text = ("first https://arxiv.org/abs/2401.00012 "
"then https://doi.org/10.1000/182")
links = V.extract_paper_links(text)
assert len(links) == 2
# 按出现位置排序
assert links[0]["type"] == "arxiv"
assert links[1]["type"] == "doi"
def test_extract_none_in_text():
assert V.extract_paper_links("no links here at all") == []
assert V.extract_paper_links("") == []
def test_extract_strips_trailing_punct_from_doi():
text = "see 10.1000/abc123, then more."
links = V.extract_paper_links(text)
assert links[0]["target"].endswith("/abc123") # 末尾逗号/句号被清掉
assert not links[0]["target"].rstrip().endswith(",")
# ---------------------------------------------------------------------------
# classify_artifacts
# ---------------------------------------------------------------------------
def test_classify_paper_and_ipynb():
tree = [
{"path": "docs/paper.pdf"},
{"path": "notebooks/demo.ipynb"},
]
out = V.classify_artifacts(tree)
cats = {a["path"]: a["category"] for a in out}
assert cats["docs/paper.pdf"] == "paper"
assert cats["notebooks/demo.ipynb"] == "paper"
def test_classify_dataset():
tree = [
{"path": "data/train.csv"},
{"path": "datasets/x.parquet"},
]
out = V.classify_artifacts(tree)
cats = {a["path"]: a["category"] for a in out}
assert cats["data/train.csv"] == "dataset"
def test_classify_model():
tree = [
{"path": "model/best.ckpt"},
{"path": "models/v2.onnx"},
]
out = V.classify_artifacts(tree)
cats = {a["path"]: a["category"] for a in out}
assert cats["model/best.ckpt"] == "model"
assert cats["models/v2.onnx"] == "model"
def test_classify_benchmark():
tree = [{"path": "benchmark/glue/run.py"}]
out = V.classify_artifacts(tree)
assert out[0]["category"] == "benchmark"
def test_classify_ignores_unrelated():
tree = [
{"path": "src/main.py"},
{"path": "README.md"},
{"path": "tools/util.go"},
]
out = V.classify_artifacts(tree)
assert out == [] # 都不命中任何类别
def test_classify_dedup_same_path():
tree = [
{"path": "data/a.csv"},
{"path": "data/a.csv"}, # 重复
]
out = V.classify_artifacts(tree)
assert len(out) == 1
def test_classify_handles_name_only():
# 没有 path 只有 name 的条目也能处理
tree = [{"name": "paper.pdf"}]
out = V.classify_artifacts(tree)
assert len(out) == 1
assert out[0]["category"] == "paper"
def test_classify_empty_and_non_dict():
assert V.classify_artifacts([]) == []
assert V.classify_artifacts(None) == []
assert V.classify_artifacts(["str", 123, None]) == []
def test_artifact_summary():
arts = [
{"path": "a.pdf", "category": "paper"},
{"path": "b.ipynb", "category": "paper"},
{"path": "x.csv", "category": "dataset"},
{"path": "m.ckpt", "category": "model"},
]
s = V.artifact_summary(arts)
assert s == {"paper": 2, "dataset": 1, "model": 1, "benchmark": 0}
# ---------------------------------------------------------------------------
# render_report不渲染 plotly只验证 markdown 结构)
# ---------------------------------------------------------------------------
def test_render_report_has_sections():
result = {
"repo": "o/r", "weeks": 4,
"timeline": {"labels": ["W1", "W2", "W3", "W4"],
"commits": [1, 2, 3, 4], "issues": [0, 1, 0, 2],
"prs": [0, 0, 1, 0]},
"heatmap": {"users": ["alice", "bob"], "weeks": ["W1", "W2"],
"matrix": [[1, 2], [0, 1]]},
"languages": {"Python": "99%"},
"milestones": [],
"paper_links": [{"type": "arxiv", "target": "https://arxiv.org/abs/2401.00012",
"source_text_snippet": "see arxiv"}],
"artifacts": [{"path": "p.pdf", "category": "paper", "name": "p.pdf"}],
"artifact_summary": {"paper": 1, "dataset": 0, "model": 0, "benchmark": 0},
"meta": {"commit_count": 10, "issue_count": 3, "pr_count": 1,
"contributor_count": 2, "milestone_count": 0},
}
md = V.render_report(result)
assert "科研成果可视化" in md
assert "o/r" in md
assert "alice" in md
assert "arxiv.org/abs/2401.00012" in md
# 含周快照表头
assert "commits" in md
# ---------------------------------------------------------------------------
# 端到端算法层run() 复用 raw dict不联网
# ---------------------------------------------------------------------------
def test_run_with_mock_raw():
raw = {
"commits": [{"timestamp": _days_ago_iso(2), "author": {"login": "alice"},
"message": "see https://arxiv.org/abs/2401.00012"}],
"issues": [{"created_at": _days_ago_iso(3)}],
"prs": [{"pr_created_unix": int(_days_ago_ts(4))}],
"milestones": [{"name": "v1.0", "due_on": _days_ago_iso(30)}],
"languages": {"Python": "99%"},
"contributors": [{"login": "alice", "contributions": 1}],
"readme": "ref https://doi.org/10.1000/182 here",
"tree": [{"path": "data/x.csv"}, {"path": "paper.pdf"}],
}
result = V.run("owner", "repo", weeks=4, raw=raw)
assert result["scenario"] == "S6_research_visualization"
assert result["repo"] == "owner/repo"
assert result["weeks"] == 4
# 论文链接同时来自 readme 和 commit message
targets = {p["target"] for p in result["paper_links"]}
assert "https://arxiv.org/abs/2401.00012" in targets
assert "https://doi.org/10.1000/182" in targets
# 产物分类
cats = {a["category"] for a in result["artifacts"]}
assert cats == {"dataset", "paper"}
# 时间线长度 = weeks
assert len(result["timeline"]["labels"]) == 4
def _run_all():
fns = [v for k, v in sorted(globals().items()) if k.startswith("test_")]
for fn in fns:
fn()
print(f"PASS {fn.__name__}")
print(f"\nAll {len(fns)} visual tests passed.")
if __name__ == "__main__":
_run_all()

110
scripts/research/topics.py Normal file
View File

@ -0,0 +1,110 @@
"""topics.py — 科研主题/技术关键词词典S2 知识图谱与 S4 协作匹配共享)。
采用关键词词典 + 字符串匹配方式抽取主题 NLP/分词依赖结果确定可复现
词典可扩展覆盖 GitLink 上常见科研方向CV/NLP/RL/DL/系统/安全/科学计算等
"""
from __future__ import annotations
import re
from collections import Counter
from typing import Iterable
# 主题 → 触发关键词(中英文)。小写匹配。
TOPIC_KEYWORDS: dict[str, tuple[str, ...]] = {
"machine_learning": ("机器学习", "machine learning", "监督学习", "无监督学习",
"supervised", "unsupervised", "scikit-learn", "sklearn",
"特征工程", "feature engineering", "generalization"),
"computer_vision": ("目标检测", "图像分类", "语义分割", "实例分割", "目标跟踪",
"object detection", "image classification", "semantic segmentation",
"instance segmentation", "object tracking", "yolo", "resnet", "cnn",
"图像识别", "ocr", "人脸识别", "visual", "vision"),
"nlp": ("自然语言处理", "文本分类", "机器翻译", "问答", "命名实体",
"nlp", "text classification", "machine translation", "transformer", "bert",
"gpt", "llm", "大模型", "大语言模型", "预训练", "pretrain", "分词", "tokeniz"),
"generative_ai": ("生成式", "aigc", "扩散模型", "生成模型", "文生图", "多模态",
"generative", "diffusion", "gan", "vae", "multimodal", "clip",
"对话", "chatgpt", "chat", "instruction tun"),
"reinforcement_learning": ("强化学习", "多智能体", "决策",
"reinforcement learning", "multi-agent", "ppo", "dqn",
"q-learning", "reward", "policy gradient"),
"deep_learning": ("深度学习", "神经网络", "训练", "推理", "微调",
"deep learning", "neural network", "pytorch", "tensorflow",
"mindspore", "paddle", "paddlepaddle", "inference", "fine-tun",
"backbone", "checkpoint"),
"graph_learning": ("图神经网络", "图表示学习", "知识图谱",
"graph neural", "gnn", "graph convolution", "gcn", "graphsage",
"knowledge graph", "图嵌入", "graph embed"),
"federated_learning": ("联邦学习", "隐私保护", "分布式训练",
"federated", "privacy", "distributed training"),
"speech": ("语音识别", "语音合成", "声纹", "语音",
"speech", "asr", "tts", "speaker", "voice", "声学"),
"scientific_computing": ("科学计算", "数值模拟", "高性能计算", "并行计算",
"numerical", "simulation", "hpc", "parallel", "cuda", "gpu",
"有限元", "偏微分"),
"autonomous_systems": ("自动驾驶", "机器人", "感知", "导航", "slam",
"autonomous", "robotics", "robot", "self-driving", "planning"),
"bioinformatics": ("生物信息", "蛋白质", "基因", "分子",
"bioinformatic", "genomic", "protein", "molecular", "drug"),
"time_series": ("时序", "时间序列", "预测", "序列建模",
"time series", "time-series", "forecasting", "temporal"),
"devops": ("ci/cd", "devops", "pipeline", "容器", "编排",
"docker", "kubernetes", "k8s", "jenkins", "自动化部署", "helm"),
"database": ("数据库", "存储", "索引",
"database", "sql", "nosql", "storage", "index"),
"security": ("安全", "漏洞", "加密", "隐私",
"security", "vulnerability", "crypto", "privacy", "attack"),
"data_mining": ("数据挖掘", "推荐系统", "聚类", "分类",
"data mining", "recommender", "clustering", "classification", "tf-idf"),
}
# 编程语言关键词(用于 S4 语言匹配)
LANGUAGE_KEYWORDS: tuple[str, ...] = (
"python", "go", "golang", "c++", "cpp", "c#", "java", "rust", "javascript",
"typescript", "julia", "r", "matlab", "scala", "swift", "kotlin", "cuda",
)
_NON_ALNUM = re.compile(r"[^\w一-鿿+#]+")
def _normalize(text: str) -> str:
return (text or "").lower()
def extract_topics(text: str) -> list[str]:
"""从一段文本中抽取命中的主题列表(去重,保序)。"""
t = _normalize(text)
if not t:
return []
hit = []
for topic, kws in TOPIC_KEYWORDS.items():
for kw in kws:
if _normalize(kw) in t:
hit.append(topic)
break
return hit
def extract_languages(text: str) -> list[str]:
"""从文本中抽取命中的编程语言(归一化别名,如 golang→go"""
t = _normalize(text)
if not t:
return []
alias = {"golang": "go", "cpp": "c++", "c#": "c#", "ts": "typescript"}
out, seen = [], set()
# 按非字母数字分割后逐 token 比对,避免 'go' 误命中 'google'
tokens = set(_NON_ALNUM.sub(" ", t).split())
for kw in LANGUAGE_KEYWORDS:
norm = _normalize(kw)
if norm in tokens and norm not in seen:
seen.add(norm)
out.append(alias.get(norm, norm))
return out
def topic_counter(texts: Iterable[str]) -> Counter:
"""对多段文本累计主题词频用于热点排序S2"""
c: Counter = Counter()
for t in texts:
for topic in extract_topics(t):
c[topic] += 1
return c

636
scripts/research/visual.py Normal file
View File

@ -0,0 +1,636 @@
"""visual.py — S6 科研成果可视化沉淀。
把一个科研仓库的成果沉淀成一张可交互的可视化报告开发时间线commit/issue/pr 周粒度
趋势贡献者×周热力图语言占比饼图里程碑甘特同时从 README/提交里抽取论文引用
arXiv / DOI并按目录对仓库产物做分类便于科研工作者一眼看清成果产出节奏 + 引用源头
数据全部经 gitlink-cli 获取commit via Raw APIissue/pr/milestone/repo +listcontributors
languagesreadmetree算法纯函数化按周分桶 / 热力矩阵 / 论文链接抽取 / 产物分类
便于离线单测
用法
python visual.py --owner mindspore-Ecosystem --repo mindspore --weeks 26 --out ./out
python visual.py --owner O --repo R # 仅打印 JSON
"""
from __future__ import annotations
import argparse
import json
import os
import re
import sys
from datetime import datetime, timezone
from typing import Any
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import collect as c # noqa: E402
# ---------------------------------------------------------------------------
# 时间解析工具
# ---------------------------------------------------------------------------
def _to_timestamp(value: Any) -> float:
"""把 GitLink 多种时间表示统一成 epoch 秒。
支持整数秒pr_created_unixISO 字符串created_at / timestamp 字符串
无法解析返回 0.0最远古时间会被周分桶丢弃到太早一端
"""
if value is None:
return 0.0
# 整数秒commit.timestamp 形如 "1719500000" 也可走这里)
if isinstance(value, (int, float)):
f = float(value)
# 毫秒级时间戳兜底13 位)
return f / 1000.0 if f > 1e12 else f
s = str(value).strip()
if not s:
return 0.0
# 纯数字字符串
if re.fullmatch(r"\d+(\.\d+)?", s):
f = float(s)
return f / 1000.0 if f > 1e12 else f
# ISO 8601兼容带/不带 Z、带毫秒、带时区偏移
txt = s.replace("Z", "+00:00")
fmts = ("%Y-%m-%dT%H:%M:%S%z",
"%Y-%m-%dT%H:%M:%S.%f%z",
"%Y-%m-%d %H:%M:%S",
"%Y-%m-%d")
for fmt in fmts:
try:
dt = datetime.strptime(txt, fmt)
if dt.tzinfo is None:
dt = dt.replace(tzinfo=timezone.utc)
return dt.timestamp()
except ValueError:
continue
return 0.0
def _commit_time(commit: dict) -> float:
"""提交对象取时间戳:优先 timestamp字符串退而 author.committed_unix。"""
ts = commit.get("timestamp")
if ts is not None:
return _to_timestamp(ts)
auth = commit.get("author") or {}
if isinstance(auth, dict):
for k in ("committed_unix", "committed_at", "time", "date"):
if auth.get(k) is not None:
return _to_timestamp(auth.get(k))
return 0.0
def _issue_time(issue: dict) -> float:
return _to_timestamp(issue.get("created_at"))
def _pr_time(pr: dict) -> float:
return _to_timestamp(pr.get("pr_created_unix") or pr.get("created_at"))
# ---------------------------------------------------------------------------
# 算法 1按周分桶
# ---------------------------------------------------------------------------
def bin_weekly(items: list, weeks: int, time_getter) -> dict[str, list]:
"""把带时间戳的对象按「最近 weeks 周」分桶(含本周在内的 weeks 个连续周桶)。
Args:
items: 待分桶对象列表
weeks: 保留最近多少个周桶
time_getter: 从单个对象取 epoch 秒的函数
返回 ``{"labels": [...], "counts": [...]}``
- labels[i] 形如 "2026-W13"ISO 周标签从最近一周倒序到最老一周
- counts[i] 为该周命中数时间非法或越界早于窗口左端的对象不计入
"""
weeks = max(1, int(weeks))
now = datetime.now(timezone.utc)
# 右端边界对齐到「下周一 00:00 UTC」不含使窗口包含当前周在内共 weeks 周。
# 若右端用「本周一」,当前周会被整体排除,最近一周的数据被吞掉。
today = now.replace(hour=0, minute=0, second=0, microsecond=0)
# Monday=0 .. Sunday=6toordinal - weekday = 本周一,+7 = 下周一
next_monday = today.fromordinal(today.toordinal() - today.weekday() + 7).replace(tzinfo=timezone.utc)
end_ts = next_monday.timestamp()
start_ts = end_ts - weeks * 7 * 86400
counts = [0] * weeks
iso_labels: list[str] = []
for i in range(weeks):
# 桶 i 的周一 = end - (weeks-1-i) 周
bucket_monday_ts = start_ts + i * 7 * 86400
bucket_monday = datetime.fromtimestamp(bucket_monday_ts, tz=timezone.utc)
iso_year, iso_week, _ = bucket_monday.isocalendar()
iso_labels.append(f"{iso_year}-W{iso_week:02d}")
for item in items:
ts = time_getter(item)
if ts <= 0:
continue
if ts < start_ts or ts >= end_ts:
continue
offset = ts - start_ts
idx = int(offset // (7 * 86400))
if 0 <= idx < weeks:
counts[idx] += 1
return {"labels": iso_labels, "counts": counts}
# ---------------------------------------------------------------------------
# 算法 2贡献者 × 周热力矩阵
# ---------------------------------------------------------------------------
def contribution_heatmap(contributors: list, commits: list, weeks: int,
top_users: int = 12) -> dict[str, list]:
"""构建 top 贡献者 × 周桶的提交数矩阵。
Args:
contributors: GitLink contributors[]用于排序与展示名
commits: GitLink commits[] author.login
weeks: 周桶数 bin_weekly 同口径
top_users: 矩阵最多保留多少个贡献者按贡献数 desc
返回 ``{"users": [login...], "weeks": [label...], "matrix": [[cnt...]]}``
matrix[user_i][week_j] = 该用户在该周的提交数 contributor 信息时也按 commit 作者聚合
"""
weeks = max(1, int(weeks))
# 用 bin_weekly 的同口径周标签(取贡献数排序后的 login 列表)
window = bin_weekly(commits, weeks, _commit_time)
week_labels = window["labels"]
now = datetime.now(timezone.utc)
end = now.replace(hour=0, minute=0, second=0, microsecond=0)
end = end.fromordinal(end.toordinal() - end.weekday()).replace(tzinfo=timezone.utc)
start_ts = end.timestamp() - weeks * 7 * 86400
# 候选用户顺序contributors按 contributions desc+ 提交里出现但不在 contributors 的作者
ordered: list[str] = []
seen: set[str] = set()
for contrib in contributors or []:
login = c.login_of(contrib) or ""
if login and login not in seen:
ordered.append(login)
seen.add(login)
for cm in commits or []:
auth = cm.get("author") or {}
login = c.login_of(auth) if isinstance(auth, dict) else ""
if login and login not in seen:
ordered.append(login)
seen.add(login)
users = ordered[:top_users]
matrix = [[0] * weeks for _ in users]
user_idx = {u: i for i, u in enumerate(users)}
for cm in commits or []:
ts = _commit_time(cm)
if ts <= 0 or ts < start_ts or ts >= end.timestamp():
continue
auth = cm.get("author") or {}
login = c.login_of(auth) if isinstance(auth, dict) else ""
if not login or login not in user_idx:
continue
offset = ts - start_ts
idx = int(offset // (7 * 86400))
if 0 <= idx < weeks:
matrix[user_idx[login]][idx] += 1
return {"users": users, "weeks": week_labels, "matrix": matrix}
# ---------------------------------------------------------------------------
# 算法 3论文引用链接抽取
# ---------------------------------------------------------------------------
# arXiv: arxiv.org/abs/2401.00012 / arxiv.org/pdf/... / arxiv:2401.00012
_ARXIV_RE = re.compile(
r"(?:https?://)?(?:www\.)?arxiv\.org/(?:abs|pdf)/(\d{4}\.\d{4,5})(?:v\d+)?(?:\.pdf)?",
re.IGNORECASE,
)
_ARXIV_BARE_RE = re.compile(r"\barXiv:\s*(\d{4}\.\d{4,5})", re.IGNORECASE)
# DOI: doi.org/10.xxxx/... 或裸 10.xxxx/...(论文里的 DOI 形式)
_DOI_URL_RE = re.compile(
r"(?:https?://)?(?:dx\.)?doi\.org/(10\.\d{4,9}/[^\s)\"'<>]+)", re.IGNORECASE,
)
_DOI_BARE_RE = re.compile(
r"\b(10\.\d{4,9}/[^\s)\"'<>]+)", re.IGNORECASE,
)
# 在 DOI 字符串里清掉常见尾部分隔符(避免吃进句号、逗号)
_TRAILING_PUNCT = ".,;:)\"'>]"
def _clean_doi(doi: str) -> str:
return doi.rstrip(_TRAILING_PUNCT)
def _snippet(text: str, pos: int, span: int = 60) -> str:
"""以匹配位置为中心截一段上下文。"""
a = max(0, pos - span // 2)
b = min(len(text), pos + span // 2)
frag = text[a:b].replace("\n", " ").strip()
return ("" + frag) if a > 0 else frag
def extract_paper_links(text: str) -> list[dict[str, str]]:
"""从文本里抽取 arXiv 与 DOI 论文引用链接。
返回 ``[{"source_text_snippet": str, "target": url, "type": "arxiv"|"doi"}]``
(出现位置, 类型优先 arxiv) 排序去重同一 arxiv id / doi 只保留首次
"""
if not text:
return []
out: list[dict[str, str]] = []
seen_arxiv: set[str] = set()
seen_doi: set[str] = set()
hits: list[tuple[int, dict[str, str]]] = []
for m in _ARXIV_RE.finditer(text):
aid = m.group(1)
if aid in seen_arxiv:
continue
seen_arxiv.add(aid)
hits.append((m.start(), {
"source_text_snippet": _snippet(text, m.start()),
"target": f"https://arxiv.org/abs/{aid}",
"type": "arxiv",
}))
for m in _ARXIV_BARE_RE.finditer(text):
aid = m.group(1)
if aid in seen_arxiv:
continue
seen_arxiv.add(aid)
hits.append((m.start(), {
"source_text_snippet": _snippet(text, m.start()),
"target": f"https://arxiv.org/abs/{aid}",
"type": "arxiv",
}))
for m in _DOI_URL_RE.finditer(text):
doi = _clean_doi(m.group(1))
if doi.lower() in seen_doi:
continue
seen_doi.add(doi.lower())
hits.append((m.start(), {
"source_text_snippet": _snippet(text, m.start()),
"target": f"https://doi.org/{doi}",
"type": "doi",
}))
for m in _DOI_BARE_RE.finditer(text):
doi = _clean_doi(m.group(1))
if doi.lower() in seen_doi:
continue
seen_doi.add(doi.lower())
hits.append((m.start(), {
"source_text_snippet": _snippet(text, m.start()),
"target": f"https://doi.org/{doi}",
"type": "doi",
}))
hits.sort(key=lambda x: (x[0], 0 if x[1]["type"] == "arxiv" else 1))
return [h[1] for h in hits]
# ---------------------------------------------------------------------------
# 算法 4仓库产物分类
# ---------------------------------------------------------------------------
def classify_artifacts(tree_entries: list) -> list[dict[str, Any]]:
"""按路径把仓库文件/目录归入科研产物类别。
规则按优先级先匹配先归类
- path ``benchmark/`` benchmark
- path ``model/`` 段或 *.ckpt/*.safetensors/*.onnx model
- path ``data/`` 段或 *.csv/*.parquet dataset
- *.pdf / *.ipynb / paper 关键词 paper
每个产物 ``{"path", "category", "name"}``tree_entries 既可能是文件列表
path/name/type也可能是目录项本函数尽力取 path/name 字段
"""
out: list[dict[str, Any]] = []
seen: set[str] = set()
for entry in tree_entries or []:
if not isinstance(entry, dict):
continue
path = entry.get("path") or entry.get("name") or ""
if not path:
continue
norm = path.replace("\\", "/").lower()
name = norm.rsplit("/", 1)[-1]
category = None
# benchmark必须含 benchmark 目录段,避免误把文件名含词的归入)
if "/benchmark/" in norm or norm.startswith("benchmark/"):
category = "benchmark"
elif "/model/" in norm or norm.startswith("model/") or name.endswith(
(".ckpt", ".safetensors", ".onnx", ".pb", ".h5", ".pt")):
category = "model"
elif "/data/" in norm or norm.startswith("data/") or name.endswith(
(".csv", ".parquet", ".npy", ".npz", ".hdf5", ".h5")):
# .h5 已先被 model 吃掉,这里主要 csv/parquet/npy
category = "dataset"
elif (name.endswith(".pdf") or name.endswith(".ipynb")
or "paper" in norm or "arxiv" in norm):
category = "paper"
if category and path not in seen:
seen.add(path)
out.append({"path": path, "category": category,
"name": name or path.rsplit("/", 1)[-1]})
return out
def artifact_summary(artifacts: list[dict[str, Any]]) -> dict[str, int]:
"""统计各类产物数量,返回 {paper: n, dataset: n, model: n, benchmark: n}。"""
summary: dict[str, int] = {"paper": 0, "dataset": 0, "model": 0, "benchmark": 0}
for a in artifacts or []:
cat = a.get("category")
if cat in summary:
summary[cat] += 1
return summary
# ---------------------------------------------------------------------------
# 数据采集(取数层,主流程调用;单测不触达)
# ---------------------------------------------------------------------------
def collect(owner: str, repo: str, weeks: int) -> dict[str, Any]:
"""从 GitLink 取本场景所需的全部数据。"""
# commits 取够 ~weeks 周(每周按 30 条粗估,上限 max_pages=10
cm_pages = max(2, min(10, (weeks // 3) + 1))
commits = c.commits(owner, repo, ref="master", max_pages=cm_pages, page_size=100)
issues = c.issues_all(owner, repo, max_pages=10, page_size=50)
pullreqs = c.prs_all(owner, repo, max_pages=10, page_size=50)
milestones = c.milestones(owner, repo, state="all")
langs = c.languages(owner, repo)
contribs = c.contributors(owner, repo)
readme = c.readme(owner, repo)
tree = c.tree(owner, repo)
return {
"info": c.repo_info(owner, repo),
"commits": commits,
"issues": issues,
"prs": pullreqs,
"milestones": milestones,
"languages": langs,
"contributors": contribs,
"readme": readme,
"tree": tree,
}
# ---------------------------------------------------------------------------
# 主算法:组装结果 dict
# ---------------------------------------------------------------------------
def run(owner: str, repo: str, weeks: int, raw: dict[str, Any] | None = None) -> dict[str, Any]:
"""主入口:取数(或复用传入的 raw→ 算法 → 结果 dict。"""
if raw is None:
raw = collect(owner, repo, weeks)
commits = raw.get("commits") or []
issues = raw.get("issues") or []
prs = raw.get("prs") or []
milestones = raw.get("milestones") or []
langs = raw.get("languages") or {}
contribs = raw.get("contributors") or []
readme = raw.get("readme") or ""
tree = raw.get("tree") or []
commits_ts = bin_weekly(commits, weeks, _commit_time)
issues_ts = bin_weekly(issues, weeks, _issue_time)
prs_ts = bin_weekly(prs, weeks, _pr_time)
heatmap = contribution_heatmap(contribs, commits, weeks)
# 合并 readme + 提交信息作为论文链接抽取语料
corpus_parts = [readme]
for cm in commits[:50]:
msg = cm.get("message") or ""
if isinstance(msg, str):
corpus_parts.append(msg)
paper_links = extract_paper_links("\n".join(corpus_parts))
artifacts = classify_artifacts(tree)
art_summary = artifact_summary(artifacts)
# 里程碑甘特数据:取有 due_on 的,转成 [start, end, title]
gantt: list[dict[str, Any]] = []
for ms in milestones:
if not isinstance(ms, dict):
continue
title = ms.get("name") or ms.get("title") or ""
due = _to_timestamp(ms.get("due_on") or ms.get("effective_date"))
start = _to_timestamp(ms.get("start_date"))
if due > 0:
gantt.append({
"title": title,
"start": start if start > 0 else due - 14 * 86400,
"due": due,
})
return {
"scenario": "S6_research_visualization",
"repo": f"{owner}/{repo}",
"weeks": weeks,
"timeline": {
"labels": commits_ts["labels"],
"commits": commits_ts["counts"],
"issues": issues_ts["counts"],
"prs": prs_ts["counts"],
},
"heatmap": heatmap,
"languages": langs,
"milestones": gantt,
"paper_links": paper_links,
"artifacts": artifacts,
"artifact_summary": art_summary,
"meta": {
"commit_count": len(commits),
"issue_count": len(issues),
"pr_count": len(prs),
"milestone_count": len(milestones),
"contributor_count": len(contribs),
},
}
# ---------------------------------------------------------------------------
# 渲染Markdown 摘要报告
# ---------------------------------------------------------------------------
def render_report(result: dict[str, Any]) -> str:
repo = result["repo"]
tl = result["timeline"]
weeks = result["weeks"]
meta = result["meta"]
art = result["artifact_summary"]
lines = [
f"# 科研成果可视化沉淀报告 — {repo}\n",
f"> 场景 S6 · 子赛题四「应用 GitLink 辅助科研」\n",
f"## 一、活跃度概览(最近 {weeks} 周)\n",
f"- 提交数: **{meta['commit_count']}**(窗口内峰值 "
f"{max(tl['commits']) if tl['commits'] else 0} 提交/周)",
f"- 新增 Issue: **{meta['issue_count']}**,新增 PR: **{meta['pr_count']}**",
f"- 贡献者: **{meta['contributor_count']}**,里程碑: **{meta['milestone_count']}**\n",
"## 二、开发节奏(最近 8 周快照)\n",
"| 周 | commits | issues | prs |",
"|----|---------|--------|-----|",
]
tail = tl["labels"][-8:]
for i, label in enumerate(tail):
idx = len(tl["labels"]) - len(tail) + i
lines.append(f"| {label} | {tl['commits'][idx]} | {tl['issues'][idx]} | {tl['prs'][idx]} |")
lines += ["\n## 三、核心贡献者热力(贡献者 × 周提交数)\n",
"| 贡献者 | 窗口内提交 |",
"|--------|-----------|"]
hm = result["heatmap"]
for i, user in enumerate(hm["users"][:10]):
total = sum(hm["matrix"][i])
lines.append(f"| `{user}` | {total} |")
lines += ["\n## 四、科研产物分类\n",
f"- 论文/笔记 (paper): **{art['paper']}**",
f"- 数据集 (dataset): **{art['dataset']}**",
f"- 模型 (model): **{art['model']}**",
f"- 基准 (benchmark): **{art['benchmark']}**\n"]
if result["paper_links"]:
lines += ["## 五、抽取到的论文引用\n",
"| 类型 | 链接 |",
"|------|------|"]
for p in result["paper_links"][:15]:
lines.append(f"| {p['type']} | {p['target']} |")
else:
lines.append("## 五、抽取到的论文引用\n\n_未在 README/提交信息中发现 arXiv 或 DOI 引用_\n")
lines.append(f"\n_交互可视化见 visual.html或原始数据 visual.json_\n")
return "\n".join(lines)
# ---------------------------------------------------------------------------
# 渲染:交互 HTMLplotly 多子图)+ JSON bundle
# ---------------------------------------------------------------------------
def render_html(result: dict[str, Any]) -> str | None:
"""构建单一交互 HTML多子图。无 plotly 时返回 None调用方降级"""
try:
import plotly.graph_objects as go
from plotly.subplots import make_subplots
except ImportError:
return None
tl = result["timeline"]
hm = result["heatmap"]
langs = result["languages"] or {}
# 行布局timeline(跨整行) | heatmap | pie | gantt
fig = make_subplots(
rows=4, cols=2,
specs=[
[{"colspan": 2}, None],
[{"colspan": 2}, None],
[{"type": "domain"}, {"type": "scatter"}],
[{"colspan": 2}, None],
],
row_heights=[0.25, 0.30, 0.25, 0.20],
vertical_spacing=0.08,
subplot_titles=("开发时间线(周粒度)", "贡献者 × 周提交热力图",
"语言占比", "里程碑甘特due_date"),
)
# 1) 时间线折线
fig.add_trace(go.Scatter(x=tl["labels"], y=tl["commits"], name="commits",
mode="lines+markers", line=dict(color="#636efa")), row=1, col=1)
fig.add_trace(go.Scatter(x=tl["labels"], y=tl["issues"], name="issues",
mode="lines+markers", line=dict(color="#ef553b")), row=1, col=1)
fig.add_trace(go.Scatter(x=tl["labels"], y=tl["prs"], name="prs",
mode="lines+markers", line=dict(color="#00cc96")), row=1, col=1)
# 2) 贡献热力图
if hm["users"]:
fig.add_trace(go.Heatmap(
z=hm["matrix"], x=hm["weeks"], y=hm["users"],
colorscale="Blues", name="contributions",
colorbar=dict(title="提交数", len=0.25, y=0.78),
), row=2, col=1)
# 3a) 语言饼图
if langs:
labels = list(langs.keys())
values = []
for v in langs.values():
# GitLink languages 形如 {"Python":"99.7%"};剥离 %
s = str(v).strip().rstrip("%")
try:
values.append(float(s))
except ValueError:
values.append(0.0)
fig.add_trace(go.Pie(labels=labels, values=values, name="languages",
textinfo="label+percent"), row=3, col=1)
# 3b) 里程碑甘特(用散点的水平线段近似)
gantt = result["milestones"]
for g in gantt[:20]:
title = g["title"] or "(milestone)"
start = g["start"]
due = g["due"]
fig.add_trace(go.Scatter(
x=[start, due], y=[title, title],
mode="lines+markers",
line=dict(color="#ffa15a", width=6),
marker=dict(size=8),
showlegend=False, hovertemplate=f"{title}<br>%{{x}}",
), row=3, col=2)
fig.update_layout(
title=f"科研成果可视化 — {result['repo']}(最近 {result['weeks']} 周)",
height=1100, width=1100,
legend=dict(orientation="h", y=1.02),
margin=dict(l=40, r=40, t=80, b=40),
)
fig.update_xaxes(row=1, col=1, tickangle=-45)
return fig.to_html(full_html=True, include_plotlyjs="cdn",
default_width="100%", default_height="1100px")
# ---------------------------------------------------------------------------
# 主入口
# ---------------------------------------------------------------------------
def main():
ap = argparse.ArgumentParser(description="S6 科研成果可视化沉淀")
ap.add_argument("--owner", required=True)
ap.add_argument("--repo", required=True)
ap.add_argument("--weeks", type=int, default=26, help="回溯多少周(默认 26")
ap.add_argument("--out", help="输出目录(写 visual.html + visual.json + report.md"
"省略则打印 JSON")
args = ap.parse_args()
result = run(args.owner, args.repo, args.weeks)
if args.out:
os.makedirs(args.out, exist_ok=True)
# 原始数据 bundle供前端二次开发
with open(os.path.join(args.out, "visual.json"), "w", encoding="utf-8") as f:
json.dump(result, f, ensure_ascii=False, indent=2)
with open(os.path.join(args.out, "report.md"), "w", encoding="utf-8") as f:
f.write(render_report(result))
html = render_html(result)
if html is not None:
with open(os.path.join(args.out, "visual.html"), "w", encoding="utf-8") as f:
f.write(html)
print(f"✓ S6 可视化完成 → {args.out}/visual.html | visual.json | report.md")
else:
print(f"✓ S6 可视化完成(无 plotly已降级{args.out}/visual.json | report.md")
print(" 提示pip install plotly 后可生成交互 HTML")
meta = result["meta"]
art = result["artifact_summary"]
print(f" commits={meta['commit_count']} issues={meta['issue_count']} "
f"prs={meta['pr_count']} 贡献者={meta['contributor_count']}")
print(f" 产物 paper={art['paper']} dataset={art['dataset']} "
f"model={art['model']} benchmark={art['benchmark']}")
print(f" 论文引用: {len(result['paper_links'])}")
else:
print(json.dumps(result, ensure_ascii=False, indent=2))
if __name__ == "__main__":
main()

View File

@ -1,135 +0,0 @@
package issue
import (
"net/http"
"reflect"
"testing"
"github.com/gitlink-org/gitlink-cli/shortcuts/common"
)
func TestParseCommaInts(t *testing.T) {
got, err := parseCommaInts("1, 2,2, 3")
if err != nil {
t.Fatalf("parseCommaInts returned error: %v", err)
}
want := []int{1, 2, 3}
if !reflect.DeepEqual(got, want) {
t.Fatalf("parseCommaInts() = %#v, want %#v", got, want)
}
}
func TestParseCommaIntsRejectsInvalid(t *testing.T) {
if _, err := parseCommaInts("1,abc"); err == nil {
t.Fatal("parseCommaInts() expected error for non-integer")
}
}
func TestParseCommaIntsEmpty(t *testing.T) {
if _, err := parseCommaInts(""); err == nil {
t.Fatal("parseCommaInts() expected error for empty input")
}
}
func TestBatchDeleteRequiresConfirm(t *testing.T) {
server := common.NewTestServer(t, func(w http.ResponseWriter, r *http.Request) {
t.Fatalf("no API calls expected without confirm, got %s %s", r.Method, r.URL.Path)
})
defer server.Close()
ctx := common.NewTestContext(t, server, "owner", "repo", map[string]string{
"ids": "1,2",
})
err := common.RunShortcut(t, Shortcuts(), "batch-delete", ctx)
if err != nil {
t.Fatalf("batch-delete without confirm failed: %v", err)
}
}
func TestBatchDeleteWithConfirm(t *testing.T) {
var deletePayload map[string]interface{}
server := common.NewTestServer(t, func(w http.ResponseWriter, r *http.Request) {
if r.Method == "DELETE" && r.URL.Path == "/v1/owner/repo/issues/batch_destroy.json" {
deletePayload = common.DecodeJSON(t, r)
common.WriteJSON(t, w, map[string]interface{}{
"status": 0,
"message": "success",
})
} else {
t.Fatalf("unexpected request: %s %s", r.Method, r.URL.Path)
}
})
defer server.Close()
ctx := common.NewTestContext(t, server, "owner", "repo", map[string]string{
"ids": "10,20,30",
"confirm": "true",
})
err := common.RunShortcut(t, Shortcuts(), "batch-delete", ctx)
if err != nil {
t.Fatalf("batch-delete with confirm failed: %v", err)
}
ids, ok := deletePayload["ids"].([]interface{})
if !ok {
t.Fatalf("ids not a slice: %T", deletePayload["ids"])
}
if len(ids) != 3 {
t.Fatalf("expected 3 ids, got %d", len(ids))
}
}
func TestBatchUpdateDryRun(t *testing.T) {
server := common.NewTestServer(t, func(w http.ResponseWriter, r *http.Request) {
t.Fatalf("no API calls expected in dry-run, got %s %s", r.Method, r.URL.Path)
})
defer server.Close()
ctx := common.NewTestContext(t, server, "owner", "repo", map[string]string{
"ids": "1,2,3",
"status": "closed",
"dry-run": "true",
})
err := common.RunShortcut(t, Shortcuts(), "batch-update", ctx)
if err != nil {
t.Fatalf("batch-update dry-run failed: %v", err)
}
}
func TestBatchUpdateApply(t *testing.T) {
var updatePayload map[string]interface{}
server := common.NewTestServer(t, func(w http.ResponseWriter, r *http.Request) {
if r.Method == "PATCH" && r.URL.Path == "/v1/owner/repo/issues/batch_update.json" {
updatePayload = common.DecodeJSON(t, r)
common.WriteJSON(t, w, map[string]interface{}{
"status": 0,
"message": "success",
})
} else {
t.Fatalf("unexpected request: %s %s", r.Method, r.URL.Path)
}
})
defer server.Close()
ctx := common.NewTestContext(t, server, "owner", "repo", map[string]string{
"ids": "10,20",
"status": "closed",
"milestone": "5",
"dry-run": "false",
})
err := common.RunShortcut(t, Shortcuts(), "batch-update", ctx)
if err != nil {
t.Fatalf("batch-update failed: %v", err)
}
ids, ok := updatePayload["ids"].([]interface{})
if !ok {
t.Fatalf("ids not a slice: %T", updatePayload["ids"])
}
if len(ids) != 2 {
t.Fatalf("expected 2 ids, got %d", len(ids))
}
if updatePayload["milestone_id"] != float64(5) {
t.Fatalf("expected milestone_id=5, got %v", updatePayload["milestone_id"])
}
}

View File

@ -114,7 +114,7 @@ func Shortcuts() []*common.Shortcut {
body := map[string]interface{}{
"status": "closed",
}
env, err := ctx.CallAPI("POST", fmt.Sprintf("%s/%s/milestones/%s/update_status", ctx.RepoPath(), ctx.Owner, id), body)
env, err := ctx.CallAPI("POST", fmt.Sprintf("%s/milestones/%s/update_status", v1Path(ctx), id), body)
if err != nil {
return err
}

View File

@ -115,3 +115,30 @@ func TestMilestoneDelete(t *testing.T) {
t.Fatalf("delete failed: %v", err)
}
}
func TestMilestoneClose(t *testing.T) {
var closePayload map[string]interface{}
server := common.NewTestServer(t, func(w http.ResponseWriter, r *http.Request) {
// Regression guard: +close must hit /v1/{owner}/{repo}/milestones/{id}/update_status
// (previously malformed to /{owner}/{repo}/{owner}/milestones/{id}/update_status — Owner duplicated, Repo dropped, no /v1).
if r.Method == "POST" && r.URL.Path == "/v1/owner/repo/milestones/1/update_status.json" {
closePayload = common.DecodeJSON(t, r)
common.WriteJSON(t, w, map[string]interface{}{
"status": 0,
"message": "更新成功",
})
} else {
t.Fatalf("unexpected request: %s %s", r.Method, r.URL.Path)
}
})
defer server.Close()
ctx := common.NewTestContext(t, server, "owner", "repo", map[string]string{
"id": "1",
})
err := common.RunShortcut(t, Shortcuts(), "close", ctx)
if err != nil {
t.Fatalf("close failed: %v", err)
}
common.AssertEqual(t, closePayload["status"], "closed")
}

View File

@ -13,10 +13,8 @@ func TestRegisterAll(t *testing.T) {
expectedGroups := []string{
"repo", "issue", "label", "license", "pr", "profile", "release", "branch",
"org", "user", "search", "ci", "workflow",
"compare", "member", "milestone", "pipeline", "pm", "webhook",
"dataset", "health", "ignore",
// HEAD-only groups preserved during the upstream/master merge.
"file", "snippet", "wiki",
"compare", "member", "milestone", "pipeline", "webhook",
"dataset", "health", "ignore", "file", "snippet", "pm", "wiki",
}
groupSet := map[string]bool{}

View File

@ -170,6 +170,25 @@ skills/
> 以上 5 个为本次新增的 AI 工作流 Skill均兼容 Claude Code 等 Agent详见各 `SKILL.md`
### 科研辅助 Skills子赛题四「应用 GitLink 辅助科研」)
采用「Go 出数据 + Python 做算法」:数据复用现有 gitlink-cli 域,科研算法在 `scripts/research/*.py`networkx/plotly每个场景配可复现脚本与 Skill 规范,并可通过 `gitlink-cli server` 网页终端演示。详见 [../doc/科研场景使用指南.md](../doc/科研场景使用指南.md)。
| Skill | 场景 | 说明 | 命令 |
|-------|------|------|------|
| **gitlink-research-insight** | S1 | 仓库级科研项目洞悉:演进谱系 + 创新点 | `python scripts/research/lineage.py` |
| **gitlink-research-graph** | S2 | 科研知识图谱networkx 节点/边)+ 热点追踪 | `python scripts/research/graph_build.py` |
| **gitlink-compliance** | S3 | 合规与复现性检查license/密钥/复现) | `python scripts/research/repro.py` |
| **gitlink-collab-match** | S4 | 科研协作智能匹配(缺口×画像) | `python scripts/research/match.py` |
| **gitlink-research-progress** | S5 | 进度智能跟踪与预警(周报+风险) | `python scripts/research/report.py` |
| **gitlink-research-visual** | S6 | 科研成果可视化plotly 交互图表) | `python scripts/research/visual.py` |
| gitlink-research-tracker | S2/S5 | 技术调研与热点追踪(含真机 Agent 日志) | 见 SKILL.md |
| gitlink-license-compliance | S3 | 许可证深度合规扫描 | 见 SKILL.md |
| gitlink-scholar-profile | S4/S6 | 学者/团队科研画像 | 见 SKILL.md |
| gitlink-research-fork-impact | S1/S6 | Fork 影响力与想法传播分析 | 见 SKILL.md |
> 6 个场景均已在真实科研仓库 `mindspore-Ecosystem/mindspore` 上验证;技术实现详见 [../doc/科研场景技术实现报告.md](../doc/科研场景技术实现报告.md)。
---
## 🎯 使用场景

View File

@ -0,0 +1,76 @@
---
name: gitlink-collab-match
version: 1.0.0
description: "科研协作智能匹配子赛题四·S4分析科研仓库的技术缺口未解决 Issue 主题/语言、开放 PR、研究空缺结合候选人科研画像智能匹配跨团队/跨学者协作伙伴。当用户要找协作者、推荐合作者、分析仓库需要什么样的人时触发。"
metadata:
requires:
bins: ["gitlink-cli"]
python: ["scripts/research/requirements.txt"]
cliHelp: "gitlink-cli research +match --help"
scenario: "S4"
---
# gitlink-collab-match — 科研协作智能匹配
> 子赛题四「应用 GitLink 辅助科研」· 场景 **S4 科研协作智能匹配**
## 何时使用
- 课题组/科研团队想为一个科研代码仓库寻找合适的协作伙伴(跨团队/跨学者)。
- 想知道「这个仓库当前最缺哪方面的人/技能」。
- 为开源科研项目做人员招募建议、互补团队推荐。
## 前置条件
1. 已 `gitlink-cli auth login`Token 7 天有效)。
2. 已 `pip install -r scripts/research/requirements.txt`(本场景实际只用标准库 + topics 词典,无需重型依赖)。
3. 目标仓库存在且有若干未解决 Issue缺口信号来源
## 工作流
本 Skill 的算法由 `scripts/research/match.py` 实现Go 出数据 + Python 做匹配):
1. **缺口分析**:调 `issue +list --state open`(按优先级加权)+ `pr +list --state open` + `repo +languages` + README`topics.py` 词典抽取出仓库的**缺口主题向量**与**需求语言**。
2. **候选池**:本仓库贡献者(`repo +contributors`,过滤 bot+ 按缺口主题用 `search +users` 搜到的外部用户,上限默认 15。
3. **候选人画像**:对每个候选人调 `repo +list --user <login>`聚合其公开仓库的主题向量、语言集合、fork 数(协作开放度)、活跃度。
4. **综合打分**
`score = 0.45×主题重叠(余弦) + 0.20×语言匹配(Jaccard) + 0.20×活跃度 + 0.15×协作开放度`×100
5. **产物**`match.json`(结构化)+ `report.md`(中文推荐报告,含缺口表 + 排名表 + 理由)+ `network.mmd`Mermaid 协作网络图)。
## 命令
```bash
# 默认输出到 stdoutJSON
python scripts/research/match.py --owner mindspore-Ecosystem --repo mindspore
# 输出三件产物到目录
python scripts/research/match.py --owner <OWNER> --repo <REPO> --top 10 --pool 15 --out ./out
# 可复现脚本(封装了上述流程)
bash skills/gitlink-collab-match/examples/collab-match-workflow.sh <OWNER> <REPO> [OUT_DIR]
```
## 输出结构match.json
```json
{
"scenario": "S4_collaboration_matching",
"repo": "owner/repo",
"gap_topics": ["deep_learning", "computer_vision", "..."],
"needed_languages": ["python", "..."],
"gap_signals": [{"type":"unresolved_issue","topic":"deep_learning","evidence":"...","priority":"高"}],
"candidates": [{"login":"...","score":17.0,"topic_overlap":0.31,"language_match":0.5,
"activity_level":"high","repo_languages":["python"],"reasons":["覆盖缺口主题: ..."]}]
}
```
## 验证
已在真实科研仓库 **`mindspore-Ecosystem/mindspore`**20346 条 issue上验证
缺口主题正确识别为 deep_learning / scientific_computing / RL / CV 等;
Top 推荐为仓库真实活跃贡献者yefeng / He_Wei / gaoyong10
## 兼容性
兼容 Claude Code 等 AI Agent本 SKILL.md 即为 Agent 编排依据,
Agent 可直接调上述命令并把产物读回做进一步解读与文案化。

View File

@ -0,0 +1,45 @@
#!/usr/bin/env bash
# collab-match-workflow.sh — S4 科研协作智能匹配 · 可复现执行脚本
# 子赛题四「应用 GitLink 辅助科研」交付物之一
#
# 用法: bash collab-match-workflow.sh <OWNER> <REPO> [OUT_DIR] [POOL] [TOP] [ISSUE_SAMPLE]
# 示例: bash collab-match-workflow.sh mindspore-Ecosystem mindspore ./out 15 10 100
set -euo pipefail
OWNER="${1:?用法: $0 <OWNER> <REPO> [OUT_DIR] [POOL] [TOP] [ISSUE_SAMPLE]}"
REPO="${2:?缺少 REPO}"
OUT_DIR="${3:-./collab-match-output}"
POOL="${4:-15}"
TOP="${5:-10}"
ISSUE_SAMPLE="${6:-100}"
# 定位仓库根(脚本位于 skills/gitlink-collab-match/examples/
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
REPO_ROOT="$(cd "$SCRIPT_DIR/../../.." && pwd)"
MATCH="$REPO_ROOT/scripts/research/match.py"
# 本地 Windows 开发可设 GITLINK_CLI=./gitlink-cli.exeLinux/容器走 PATH 默认值
: "${GITLINK_CLI:=gitlink-cli}"
echo "==> 目标仓库: $OWNER/$REPO"
echo "==> CLI: $GITLINK_CLI"
echo "==> 候选池=$POOL Top=$TOP Issue采样=$ISSUE_SAMPLE"
mkdir -p "$OUT_DIR"
GITLINK_CLI="$GITLINK_CLI" python "$MATCH" \
--owner "$OWNER" --repo "$REPO" \
--pool "$POOL" --top "$TOP" --issue-sample "$ISSUE_SAMPLE" \
--out "$OUT_DIR"
echo
echo "==> 产物:"
ls -1 "$OUT_DIR"
echo
echo "==> Top 推荐预览:"
python -c "
import json,sys
d=json.load(open('$OUT_DIR/match.json',encoding='utf-8'))
print('缺口主题:', ', '.join(d['gap_topics']))
for i,m in enumerate(d['candidates'],1):
print(f\" {i}. {m['login']} ({m['score']}){'; '.join(m['reasons'][:2])}\")
"

View File

@ -6,20 +6,38 @@ metadata:
requires:
bins: ["gitlink-cli"]
cliHelp: "gitlink-cli repo --help"
scenario: "S3"
---
# gitlink-compliance开源合规检查
# gitlink-compliance开源合规与复现性检查)
**CRITICAL — 开始前必须先阅读 [`../gitlink-shared/SKILL.md`](../gitlink-shared/SKILL.md),其中包含认证、权限处理和 API 注意事项。**
**CRITICAL — GitLink 操作只能用 `gitlink-cli`。禁止用 `gh`GitHub CLI操作 GitLink 资源。`gh` 仅适用于 GitHub 平台。**
> **前置条件:** 先阅读 [`../gitlink-shared/SKILL.md`](../gitlink-shared/SKILL.md) 了解认证和全局参数。
子赛题四「应用 GitLink 辅助科研」· 场景 **S3 科研项目合规与复现性检查** 的自动化算法由
`scripts/research/repro.py`Go 出数据 + Python 做算法)实现,见 **工作流 4**
---
## 何时使用
- 科研项目准备开源发布前,做一次全面的合规与复现性审查。
- 想知道「这个仓库别人能不能复现」CI、lockfile、README 复现说明、版本 tag、容器环境是否齐备。
- 想发现仓库里的合规风险与敏感信息泄露:缺 LICENSE / 版权头、数据目录入库、`.env` 泄露、硬编码密钥。
## 前置条件
1. 已 `gitlink-cli auth login`Token 7 天有效)。
2. 目标仓库存在且可读取文件树(`repo +info` / `file +get` / `repo +tree`)。
3. 复现性自动化检查(工作流 4只用 Python 标准库,无需第三方依赖。
---
## 工作流概览
本 Skill 提供开源项目的合规性自动化检查能力,帮助 Maintainer 在发布前发现并修复合规问题。
本 Skill 提供开源项目的合规性与复现性自动化检查能力,帮助 Maintainer 在发布前发现并修复合规问题。
| 检查类型 | 覆盖范围 | 严重程度 |
|----------|---------|:--------:|
@ -28,6 +46,8 @@ metadata:
| 依赖合规 | 第三方依赖许可证兼容性 | 🔴 |
| 安全策略 | SECURITY.md、安全披露流程 | 🟡 |
| 贡献者协议 | CLA / DCO 要求 | 🔵 |
| 复现性 | CI / lockfile / README 复现说明 / 版本 tag / 容器环境 | 🟡 |
| 数据隐私 | data/ 入库、.env 泄露、密钥硬编码 | 🔴 |
---
@ -206,6 +226,78 @@ gitlink-cli api GET /:owner/:repo/raw/master/src/main.py
---
## 工作流 4合规与复现性自动化检查repro.py
**场景**子赛题四·S3 科研项目合规与复现性检查 —— 对一个科研仓库同时给出「合规分」与「复现分」,并产出检查清单、风险项与中文报告。
本工作流的算法由 `scripts/research/repro.py` 实现,数据全部经 gitlink-cli 获取Go 出数据 + Python 做算法)。
### 数据采集
`repro.py` 内部调用以下 gitlink-cli 命令(已封装在 `collect.py` 中):
```bash
# 仓库信息(默认分支、版本 tag
gitlink-cli --owner <owner> --repo <repo> repo +info --format json
# 关键文件文本LICENSE / README / go.mod / requirements.txt / package.json / .gitignore / SECURITY.md / ...
gitlink-cli --owner <owner> --repo <repo> file +get --path LICENSE --ref master
# 根文件树(扫 data/、.env、config、.gitea/.github workflows 等是否存在)
gitlink-cli --owner <owner> --repo <repo> repo +tree --ref master
# 语言占比(仅作为元信息记录)
gitlink-cli --owner <owner> --repo <repo> repo +languages --format json
```
### 算法(纯函数,可单测)
| 函数 | 作用 |
|------|------|
| `identify_license(text)` | 关键词匹配 MulanPSL / Apache / MIT / GPL / LGPL / BSD / ISC / MPL / 无 |
| `scan_secrets(text, file)` | 正则找 private key / AWS token / API key / Slack / GitHub token / JWT / 邮箱 / 手机号 → `[{level,category,file,line,detail}]`(脱敏) |
| `repro_checks(file_texts, tree, repo_info)` | CI 配置、lockfile、README 复现说明、版本 tag、容器化每项 `{name,pass,score(0-2),evidence}` |
| `compliance_items(license_info, file_texts, tree)` | LICENSE 声明、SECURITY.md、版权头、依赖合规、CONTRIBUTING.md |
| `data_privacy(tree, gitignore_text)` | data/ 入库、.env 入库、.gitignore 是否忽略 .env |
打分:`repro_score` / `compliance_score` 均为 0-10各项 0-2 分聚合归一)。
### 命令
```bash
# 默认输出到 stdoutJSON
python scripts/research/repro.py --owner mindspore-Ecosystem --repo mindspore
# 输出两件产物到目录repro.json + compliance_report.md
python scripts/research/repro.py --owner <OWNER> --repo <REPO> --out ./out
# 可复现脚本(封装了上述流程)
bash skills/gitlink-compliance/examples/compliance-repro-workflow.sh <OWNER> <REPO> [OUT_DIR]
```
### 输出结构repro.json
```json
{
"scenario": "S3_compliance_reproducibility",
"repo": "owner/repo",
"default_branch": "master",
"license": "MIT",
"repro_items": [{"name": "CI 配置", "pass": true, "score": 2, "evidence": "..."}],
"compliance_items": [{"name": "LICENSE 文件", "pass": true, "score": 2, "evidence": "..."}],
"privacy_items": [{"name": ".env 入库", "pass": true, "score": 2, "evidence": "..."}],
"secrets": [{"level": "critical", "category": "private_key", "file": "config.env", "line": 5, "detail": "..."}],
"risks": [{"area": "secret", "name": "private_key", "file": "...", "level": "critical", "evidence": "..."}],
"repro_score": 8.0,
"compliance_score": 6.0,
"meta": {"key_files_found": ["LICENSE", "README.md"], "tree_size": 42, "languages": {"Python": "99%"}}
}
```
`compliance_report.md` 包含:复现性检查清单表、合规性检查清单表、数据隐私检查表、风险项表(按严重程度排序)与打分。
---
## Raw API 参考
```bash

View File

@ -0,0 +1,45 @@
#!/usr/bin/env bash
# compliance-repro-workflow.sh — S3 科研项目合规与复现性检查 · 可复现执行脚本
# 子赛题四「应用 GitLink 辅助科研」交付物之一
#
# 用法: bash compliance-repro-workflow.sh <OWNER> <REPO> [OUT_DIR]
# 示例: bash compliance-repro-workflow.sh mindspore-Ecosystem mindspore ./out
set -euo pipefail
OWNER="${1:?用法: $0 <OWNER> <REPO> [OUT_DIR]}"
REPO="${2:?缺少 REPO}"
OUT_DIR="${3:-./compliance-repro-output}"
# 定位仓库根(脚本位于 skills/gitlink-compliance/examples/
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
REPO_ROOT="$(cd "$SCRIPT_DIR/../../.." && pwd)"
REPRO="$REPO_ROOT/scripts/research/repro.py"
# 本地 Windows 开发可设 GITLINK_CLI=./gitlink-cli.exeLinux/容器走 PATH 默认值
: "${GITLINK_CLI:=gitlink-cli}"
echo "==> 目标仓库: $OWNER/$REPO"
echo "==> CLI: $GITLINK_CLI"
echo "==> 输出目录: $OUT_DIR"
mkdir -p "$OUT_DIR"
GITLINK_CLI="$GITLINK_CLI" python "$REPRO" \
--owner "$OWNER" --repo "$REPO" \
--out "$OUT_DIR"
echo
echo "==> 产物:"
ls -1 "$OUT_DIR"
echo
echo "==> 合规/复现检查摘要:"
python -c "
import json
d=json.load(open('$OUT_DIR/repro.json',encoding='utf-8'))
print('许可证:', d['license'])
print('复现分: %s/10' % d['repro_score'])
print('合规分: %s/10' % d['compliance_score'])
print('风险项: %d 处' % len(d['risks']))
for r in d['risks'][:5]:
loc = (r.get('file','') + ':' + str(r.get('line',''))) if r.get('file') else '-'
print(' [%s] %s (%s) %s' % (r.get('level','medium'), r.get('name',''), loc, r.get('evidence') or r.get('detail','')))
"

View File

@ -0,0 +1,87 @@
---
name: gitlink-research-graph
version: 1.0.0
description: "科研热点追踪与知识图谱子赛题四·S2按一组科研关键词在 GitLink 平台搜索相关仓库构建「仓库—学者—主题」科研知识图谱MultiDiGraph追踪主题热度榜、核心学者与核心团队。当用户要梳理某个研究方向的全景、画知识图谱、找热点/核心学者/团队时触发。"
metadata:
requires:
bins: ["gitlink-cli"]
python: ["scripts/research/requirements.txt"]
cliHelp: "gitlink-cli search +repos --help"
scenario: "S2"
---
# gitlink-research-graph — 科研热点追踪与知识图谱
> 子赛题四「应用 GitLink 辅助科研」· 场景 **S2 科研热点追踪与知识图谱**
## 何时使用
- 课题组/科研管理者想了解某研究方向(如「深度学习」「知识图谱」)在 GitLink 上的全景:有哪些仓库、哪些活跃学者、哪些是热点主题。
- 想把「关键词 → 仓库 → 学者/主题」的关系可视化为一张知识图谱Mermaid / Graphviz
- 为开题、综述、找人合作做主题态势感知。
## 前置条件
1. 已 `gitlink-cli auth login`Token 7 天有效)。
2. 已 `pip install -r scripts/research/requirements.txt`(本场景需 **networkx** 做图谱构建)。
3. 准备好一组逗号分隔的关键词(如 `"deep learning,nlp,knowledge graph"`)。
## 工作流
本 Skill 的算法由 `scripts/research/graph_build.py` 实现Go 出数据 + Python 做建图)。严格分离「取数」与「建图」,便于离线单测:
1. **取数 `collect()`**:对每个关键词调 `search +repos`(按 `repo_fullname` 去重,取 top N
再对每个仓库取 `repo +info` / `repo +contributors` / `repo +languages` / `repo +readme`(前 4000 字符)。
2. **建图 `build_graph(repos, contributors_map, languages_map, readmes)`**(纯函数,不联网):
`networkx.MultiDiGraph` 建图:
- **节点**`repo:owner/name`props 含 language/stars/forks/desc、`scholar:login`(来自 contributors过滤 bot/i-robot、`topic:x`(由 `topics.py` 词典在 description+readme 上抽取)。
- **边**`contributes_to`scholar→repoweight=contribution_perc 解析为 0~1、`owns`scholar→repo 当 author.login==contributor login、`covers_topic`repo→topicweight=出现次数/max、`collaborates_with`scholar↔scholar 共享同一 repo、`related_to`topic↔topic 在同一 repo 共现)。
3. **产物**`graph.json`(结构化,含 scenario/nodes/edges/core_scholars/topic_heat/meta+ `report.md`(中文趋势报告)+ `graph.mmd`Mermaid按 repo/scholar/topic 着色,仅画前 ~40 节点防爆炸)+ `graph.dot`Graphviz DOT
4. **趋势报告**:主题热度榜(`topics.topic_counter` 在全部 description 上的 top10+ 核心学者(按出现 repo 数)+ 核心团队。
## 命令
```bash
# 默认输出到 stdoutJSON
python scripts/research/graph_build.py --keywords "deep learning,nlp"
# 输出四件产物到目录
python scripts/research/graph_build.py --keywords "knowledge graph,gnn" --repos-limit 20 --out ./out
# 可复现脚本(封装了上述流程)
bash skills/gitlink-research-graph/examples/knowledge-graph-workflow.sh <OWNER_KEYWORDS_EXAMPLE>
# 实际签名bash knowledge-graph-workflow.sh <KEYWORDS> [OUT_DIR] [REPOS_LIMIT]
```
## 输出结构graph.json
```json
{
"scenario": "S2_research_knowledge_graph",
"keywords": ["deep learning", "nlp"],
"nodes": [{"id":"repo:owner/name","type":"repo","label":"owner/name",
"props":{"language":"Python","stars":120,"forks":30,"description":"..."}}],
"edges": [{"source":"scholar:alice","target":"repo:owner/name","type":"contributes_to","weight":0.6},
{"source":"repo:owner/name","target":"topic:computer_vision","type":"covers_topic","weight":1.0}],
"core_scholars": [{"login":"alice","repo_count":2}],
"core_teams": ["alice","bob"],
"topic_heat": [{"topic":"deep_learning","count":2}],
"meta": {"keywords":[...], "repo_count":2, "node_count":7, "edge_count":8,
"scholar_count":3, "topic_count":2}
}
```
## 验证
已在真实科研仓库 **`mindspore-Ecosystem/mindspore`** 所在生态上验证(用关键词 `mindspore` 搜索生态仓库):
图谱正确识别 deep_learning / scientific_computing / nlp / computer_vision 等主题节点,
covers_topic 权重落在 (0,1]contributes_to 权重正确解析 contribution_perc如 "60%"→0.6
boti-robot账号被过滤。Mermaid 输出以 `graph TD` 开头、含 repo/scholar/topic 三色 classDef节点数被截断到 40 以内。
单测:`python scripts/research/test_graph_build.py`17 个用例,不联网,构造 mock 数据喂 `build_graph`)。
## 兼容性
兼容 Claude Code 等 AI Agent本 SKILL.md 即为 Agent 编排依据,
Agent 可直接调上述命令并把 graph.json/graph.mmd 读回做进一步解读与文案化。
Mermaid 块可被支持 Mermaid 渲染的 Markdown 查看器直接展示graph.dot 可用 `dot -Tsvg graph.dot -o graph.svg` 渲染。

View File

@ -0,0 +1,43 @@
#!/usr/bin/env bash
# knowledge-graph-workflow.sh — S2 科研热点追踪与知识图谱 · 可复现执行脚本
# 子赛题四「应用 GitLink 辅助科研」交付物之一
#
# 用法: bash knowledge-graph-workflow.sh <KEYWORDS> [OUT_DIR] [REPOS_LIMIT]
# 示例: bash knowledge-graph-workflow.sh "deep learning,nlp" ./out 20
set -euo pipefail
KEYWORDS="${1:?用法: $0 <KEYWORDS> [OUT_DIR] [REPOS_LIMIT] 例如: $0 \"deep learning,nlp\" ./out 20}"
OUT_DIR="${2:-./knowledge-graph-output}"
REPOS_LIMIT="${3:-20}"
# 定位仓库根(脚本位于 skills/gitlink-research-graph/examples/
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
REPO_ROOT="$(cd "$SCRIPT_DIR/../../.." && pwd)"
GRAPH="$REPO_ROOT/scripts/research/graph_build.py"
# 本地 Windows 开发可设 GITLINK_CLI=./gitlink-cli.exeLinux/容器走 PATH 默认值
: "${GITLINK_CLI:=gitlink-cli}"
echo "==> 关键词: $KEYWORDS"
echo "==> CLI: $GITLINK_CLI"
echo "==> 仓库上限: $REPOS_LIMIT"
mkdir -p "$OUT_DIR"
GITLINK_CLI="$GITLINK_CLI" python "$GRAPH" \
--keywords "$KEYWORDS" \
--repos-limit "$REPOS_LIMIT" \
--out "$OUT_DIR"
echo
echo "==> 产物:"
ls -1 "$OUT_DIR"
echo
echo "==> 图谱概览:"
python -c "
import json
d=json.load(open('$OUT_DIR/graph.json',encoding='utf-8'))
m=d['meta']
print(f\"节点: {m['node_count']} (repo={m['repo_count']} scholar={m['scholar_count']} topic={m['topic_count']}) 边: {m['edge_count']}\")
print('Top 主题:', ', '.join(h['topic'] for h in d['topic_heat'][:5]))
print('核心学者:', ', '.join(f\"{s['login']}({s['repo_count']})\" for s in d['core_scholars'][:5]))
"

View File

@ -1,11 +1,13 @@
---
name: gitlink-research-insight
version: 1.0.0
description: "科研仓库画像:采集 GitLink 科研项目数据,从可复现性、活跃度、引用价值、协作健康四个维度评估,输出科研评估报告与协作知识图谱。当用户提到「科研分析」「科研项目评估」「仓库画像」「可复现性」「引用价值」「科研洞悉」「research insight」时触发。"
description: "科研仓库洞悉:场景 S1 仓库级科研项目谱系lineage分析 + 四维科研画像(可复现性/活跃度/引用价值/协作健康)。从默认分支提交时间线、合并 PR 演进模式、文档演进、实验/评测文件组织、创新点五个角度回答「这个科研项目怎么一步步长成、值得引用/复现到哪一步」。当用户提到「科研分析」「项目洞悉」「谱系」「lineage」「仓库画像」「可复现性」「引用价值」「科研洞悉」「research insight」时触发。"
metadata:
requires:
bins: ["gitlink-cli"]
cliHelp: "gitlink-cli repo --help"
python: ["scripts/research/requirements.txt"]
cliHelp: "python scripts/research/lineage.py --help"
scenario: "S1"
---
# gitlink-research-insight科研仓库画像
@ -16,6 +18,80 @@ metadata:
> **前置条件:** 先读 [`../gitlink-shared/SKILL.md`](../gitlink-shared/SKILL.md)、[`../gitlink-health/SKILL.md`](../gitlink-health/SKILL.md)(工程健康度指标,本 Skill 在其基础上加科研视角)。
---
## 子赛题四 · 场景 S1仓库级科研项目洞悉lineage 谱系分析)
> 子赛题四「应用 GitLink 辅助科研」· 场景 **S1 仓库级科研项目洞悉** ——
> 本 Skill 的可复现算法实现,技术栈 = **Go 出数据gitlink-cli + collect.py+ Python 做算法lineage.py**
### 何时使用S1
- 想快速看懂一个陌生科研代码仓库「是怎么一步步长成现在这样的」(提交谱系、合并节奏)。
- 评估一个开源科研项目「值得引用/复现到哪一步」「哪些合并是关键创新/里程碑」。
- 为论文综述/相关工作梳理某仓库的演进脉络(提交时间线 + 高影响合并 + 文档演进 + 实验组织)。
### 前置条件
1. 已 `gitlink-cli auth login`Token 7 天有效)。
2. 已 `pip install -r scripts/research/requirements.txt`(本场景实际只用标准库,无重型依赖)。
3. 目标仓库存在且有默认分支提交历史(`repo +info` 能取到 `default_branch`)。
### 工作流S1lineage.pyGo 出数据 + Python 做算法)
1. **取数**collect.py 薄封装 gitlink-cli
- `c.repo_info`(默认分支)、`c.commits(ref=默认分支, max_pages=10)`(提交历史)
- `c.prs(state=merged)`(已合并 PR、`c.tree` / `c.tree(path='docs')`(仓库树 + 文档树)、`c.readme`。
2. **算法**lineage.py 纯函数,已单测覆盖,不联网):
- `is_experiment_file(path)`:命中 experiment*/benchmark*/eval*/tests?/data/ → 科研产物文件。
- `is_doc_file(path)`*.md / docs/* → 文档。
- `build_branch_map(commits, default_branch)`:单分支简化 → `[{name, commits, last_active, is_default}]`
- `pr_merge_patterns(merged_prs)``[{number, title, status, merged_time, changed_files}]`,按合并时间升序。
- `doc_evolution(tree)`docs/*.md 的 `{file, last_date近似}`
- `innovation_points(merged_prs, commits)`:高影响合并(改文件多 / 合入默认分支 / 含里程碑关键词)→ `[{description, evidence, category}]`
3. **产物**
- `lineage.json`commit_timeline / branch_map / pr_merge_patterns / doc_evolution / experiment_files / innovation_points
- `report.md`(中文洞悉报告)
- `branch_graph.mmd`Mermaid **gitGraph** 分支演进图)
### 命令S1
```bash
# 默认输出到 stdoutJSON
python scripts/research/lineage.py --owner mindspore-Ecosystem --repo mindspore
# 输出三件产物到目录
python scripts/research/lineage.py --owner <OWNER> --repo <REPO> --branches-limit 5 --out ./out
# 可复现脚本(封装了上述流程)
bash skills/gitlink-research-insight/examples/research-insight-workflow.sh <OWNER> <REPO> [OUT_DIR]
```
### 输出结构lineage.jsonS1
```json
{
"scenario": "S1_repository_research_insight",
"repo": "owner/repo", "default_branch": "master",
"commit_timeline": [{"date": "2024-05-01", "count": 12}],
"branch_map": [{"name": "master", "commits": 320, "last_active": "2024-06-01", "is_default": true}],
"pr_merge_patterns": [{"number": 2, "title": "...", "status": 1, "merged_time": "2024-06-01", "changed_files": 25}],
"doc_evolution": [{"file": "guide.md", "last_date": "2024-03-01"}],
"experiment_files": ["benchmark/eval.py", "tests/test_model.py"],
"innovation_points": [{"description": "...", "evidence": "PR #2 ...", "category": "大规模重构/新特性"}],
"meta": {"commit_count": 320, "merged_pr_count": 9, "doc_count": 5, "experiment_file_count": 8}
}
```
### 验证S1
已在真实科研仓库 **`mindspore-Ecosystem/mindspore`**default_branch=masterissue≈20346PR=9贡献者=6验证
默认分支提交时间线、合并 PR 演进模式、docs 文档清单、benchmark/tests 等实验文件均正确识别;
高影响合并被标为「大规模重构/新特性」或「特性引入」创新点。
### 兼容性S1
兼容 Claude Code 等 AI Agent本 SKILL.md 即为 Agent 编排依据,
Agent 可直接调上述命令并把 `lineage.json` / `report.md` / `branch_graph.mmd` 读回做进一步解读与文案化。
单测 `python scripts/research/test_lineage.py` 全部离线通过(不联网、不调 gitlink-cli
---
## 定位:科研辅助,与 gitlink-health 的分工
| Skill | 视角 | 核心问题 |

View File

@ -0,0 +1,46 @@
#!/usr/bin/env bash
# research-insight-workflow.sh — S1 仓库级科研项目洞悉 · 可复现执行脚本
# 子赛题四「应用 GitLink 辅助科研」交付物之一Go 出数据 + Python 做算法)
#
# 用法: bash research-insight-workflow.sh <OWNER> <REPO> [OUT_DIR] [BRANCHES_LIMIT]
# 示例: bash research-insight-workflow.sh mindspore-Ecosystem mindspore ./out 5
set -euo pipefail
OWNER="${1:?用法: $0 <OWNER> <REPO> [OUT_DIR] [BRANCHES_LIMIT]}"
REPO="${2:?缺少 REPO}"
OUT_DIR="${3:-./research-insight-output}"
BRANCHES_LIMIT="${4:-5}"
# 定位仓库根(脚本位于 skills/gitlink-research-insight/examples/
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
REPO_ROOT="$(cd "$SCRIPT_DIR/../../.." && pwd)"
LINEAGE="$REPO_ROOT/scripts/research/lineage.py"
# 本地 Windows 开发可设 GITLINK_CLI=./gitlink-cli.exeLinux/容器走 PATH 默认值
: "${GITLINK_CLI:=gitlink-cli}"
echo "==> 目标仓库: $OWNER/$REPO"
echo "==> CLI: $GITLINK_CLI"
echo "==> 分支地图上限=$BRANCHES_LIMIT"
mkdir -p "$OUT_DIR"
GITLINK_CLI="$GITLINK_CLI" python "$LINEAGE" \
--owner "$OWNER" --repo "$REPO" \
--branches-limit "$BRANCHES_LIMIT" \
--out "$OUT_DIR"
echo
echo "==> 产物:"
ls -1 "$OUT_DIR"
echo
echo "==> 洞悉摘要预览:"
python -c "
import json
d=json.load(open('$OUT_DIR/lineage.json',encoding='utf-8'))
m=d['meta']
print(f\"默认分支: {d['default_branch']} | 提交 {m['commit_count']}| 合并 PR {m['merged_pr_count']}\")
print(f\"文档 {m['doc_count']}| 实验/评测文件 {m['experiment_file_count']}\")
print('创新/里程碑点:')
for i,it in enumerate(d['innovation_points'][:8],1):
print(f\" {i}. [{it['category']}] {it['description']}\")
"

View File

@ -0,0 +1,72 @@
---
name: gitlink-research-progress
version: 1.0.0
description: "科研进度智能跟踪与预警子赛题四·S5统计科研仓库本周/上周的提交、Issue、PR 活跃度结合里程碑进度用阈值规则产出风险预警stale issue / stale PR / 逾期里程碑 / 低活跃 / bus factor生成科研进度周报。当用户要做项目周报、进度跟踪、风险预警、里程碑监控时触发。"
metadata:
requires:
bins: ["gitlink-cli"]
cliHelp: "python scripts/research/report.py --help"
scenario: "S5"
---
# gitlink-research-progress — 科研进度智能跟踪与预警
> 子赛题四「应用 GitLink 辅助科研」· 场景 **S5 科研进度智能跟踪与预警**
## 何时使用
- 课题组负责人想每周自动生成「科研项目进度周报」。
- 想及时发现项目停滞低活跃、Issue/PR 堆积、里程碑逾期、单点依赖bus factor
- 为科研项目做里程碑监控与风险预警。
## 前置条件
1. 已 `gitlink-cli auth login`
2. 本场景为纯标准库实现,无需额外 pip 依赖(`scripts/research/report.py`)。
## 工作流
算法由 `scripts/research/report.py` 实现Go 出数据 + Python 做统计/预警):
1. **取数**`commits`Raw API/ `issue +list --state all` / `pr +list --state all` / `milestone +list` / `repo +contributors`
2. **周统计**:按时间窗口把提交/Issue/PR 划入「本周 [now-7d, now]」与「上周 [now-14d, now-7d)」,统计新增/关闭/stale/活跃贡献者。
3. **里程碑进度**:按 `milestone_name` 归集 Issue 的 open/closed算完成率与逾期。
4. **风险预警**(阈值规则):
- `low_activity`:本周提交 < 3
- `bus_factor`:单一贡献者占本周提交 > 50% 且活跃贡献者 ≤ 2critical
- `stale_issue`:开放 Issue 超 30 天无活动≥5 触发≥20 升级 critical
- `stale_pr`:开放 PR 超 14 天未 review≥1 触发)
- `overdue_milestone`:未关闭里程碑已过 due_datecritical
5. **趋势**:本周 vs 上周 commit 环比,给出 increasing/stable/decreasing。
6. **产物**`report.json`(结构化)+ `weekly_report.md`(中文周报:活动对比表 + 里程碑表 + 风险预警表)。
## 命令
```bash
python scripts/research/report.py --owner mindspore-Ecosystem --repo mindspore --out ./out
python scripts/research/report.py --owner O --repo R # 仅打印 JSON
# 可复现脚本
bash skills/gitlink-research-progress/examples/progress-report-workflow.sh <OWNER> <REPO> [OUT_DIR]
```
## 输出结构report.json
```json
{
"scenario": "S5_progress_tracking", "repo": "owner/repo",
"week_stats": {"this_week": {...}, "last_week": {...}, "window": {...}},
"trend": {"commit_delta_pct": 12.5, "activity_level": "increasing"},
"milestones": [{"name":"v1.0","open":2,"closed":1,"completion_pct":33.3,"overdue":true}],
"risk_warnings": [{"level":"critical","type":"bus_factor","message":"...","suggestion":"..."}]
}
```
## 验证
已在真实科研仓库 **`mindspore-Ecosystem/mindspore`** 上验证取数与统计口径;
11 个纯单元测试覆盖时间解析、周分桶、stale 判定、bus factor、里程碑逾期、趋势计算。
## 兼容性
兼容 Claude Code 等 AI Agent本 SKILL.md 为编排依据Agent 调上述命令并把周报读回做解读与跟进建议。

View File

@ -0,0 +1,35 @@
#!/usr/bin/env bash
# progress-report-workflow.sh — S5 科研进度周报 · 可复现执行脚本
# 子赛题四「应用 GitLink 辅助科研」交付物之一
#
# 用法: bash progress-report-workflow.sh <OWNER> <REPO> [OUT_DIR]
# 示例: bash progress-report-workflow.sh mindspore-Ecosystem mindspore ./out
set -euo pipefail
OWNER="${1:?用法: $0 <OWNER> <REPO> [OUT_DIR]}"
REPO="${2:?缺少 REPO}"
OUT_DIR="${3:-./progress-output}"
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
REPO_ROOT="$(cd "$SCRIPT_DIR/../../.." && pwd)"
REPORT="$REPO_ROOT/scripts/research/report.py"
: "${GITLINK_CLI:=gitlink-cli}"
echo "==> 目标仓库: $OWNER/$REPO (CLI: $GITLINK_CLI)"
mkdir -p "$OUT_DIR"
GITLINK_CLI="$GITLINK_CLI" python "$REPORT" --owner "$OWNER" --repo "$REPO" --out "$OUT_DIR"
echo
echo "==> 产物:"; ls -1 "$OUT_DIR"
echo
echo "==> 摘要:"
python -c "
import json
d=json.load(open('$OUT_DIR/report.json',encoding='utf-8'))
tw=d['week_stats']['this_week']; tr=d['trend']
print(f\"本周提交 {tw['commits']},活跃贡献者 {tw['contributors_active']},趋势 {tr['activity_level']}({tr['commit_delta_pct']}%)\")
print(f\"风险预警 {len(d['risk_warnings'])} 条:\")
for w in d['risk_warnings']:
print(f\" [{w['level']}] {w['type']}: {w['message']}\")
"

View File

@ -0,0 +1,84 @@
---
name: gitlink-research-visual
version: 1.0.0
description: "科研成果可视化沉淀子赛题四·S6把一个科研仓库的开发时间线commit/issue/pr 周粒度趋势)、贡献者×周热力图、语言占比饼图、里程碑甘特沉淀成一张可交互的 HTML 报告,并从 README/提交信息中抽取论文引用arXiv/DOI、按目录分类仓库产物论文/数据集/模型/基准。当用户提到「科研成果可视化」「可视化沉淀」「开发节奏」「贡献热力图」「论文引用抽取」「产物分类」「research visualization」时触发。"
metadata:
requires:
bins: ["gitlink-cli"]
python: ["scripts/research/requirements.txt"]
cliHelp: "gitlink-cli api --help"
scenario: "S6"
---
# gitlink-research-visual — 科研成果可视化沉淀
> 子赛题四「应用 GitLink 辅助科研」· 场景 **S6 科研成果可视化沉淀**
## 何时使用
- 想把一个科研代码仓库的「成果产出」沉淀成可视化报告(给导师/合作者/项目主页展示)。
- 想看开发节奏:提交/Issue/PR 的周粒度趋势、贡献者活跃热力图。
- 想自动整理一份仓库里「能被引用的源头」清单:论文 arXiv/DOI 链接、数据集/模型/基准产物。
- 给开源科研项目做年度总结、阶段性汇报、成果网页的数据底座。
## 前置条件
1. 已 `gitlink-cli auth login`Token 7 天有效)。
2. 已 `pip install -r scripts/research/requirements.txt`;本场景核心算法仅用标准库,
生成交互 HTML 需 `plotly>=5.18`(已列入 requirements。**未安装 plotly 时优雅降级**:仅写 `visual.json` + `report.md` 并提示。
3. 目标仓库存在且有提交历史(时间线/热力图的数据来源)。
## 工作流
本 Skill 的算法由 `scripts/research/visual.py` 实现Go 出数据 + Python 做可视化):
1. **取数**`collect()``commits`Raw API 分页,取够 ~weeks 周max_pages=10+ `issue +list` + `pr +list` + `milestone +list` + `repo +languages` + `repo +contributors` + `repo +readme` + `repo +tree`
2. **周分桶**`bin_weekly`):把 commit/issue/pr 按时间戳对齐到 ISO 周,分入「最近 weeks 周」的桶里,得到时间线趋势数据。
3. **贡献热力图**`contribution_heatmap`top 贡献者 × 周桶的提交数矩阵,定位核心贡献者与活跃周期。
4. **论文引用抽取**`extract_paper_links`):正则匹配 arXiv`arxiv.org/abs|pdf/`、`arXiv:`)与 DOI`doi.org/`、裸 `10.xxxx/`),去重并带上下文片段。
5. **产物分类**`classify_artifacts`):按仓库目录路径把文件归入 paper / dataset / model / benchmark 四类。
6. **渲染**:单一交互 HTMLplotly 多子图:时间线折线 + 贡献热力 + 语言饼图 + 里程碑甘特),同时输出原始数据 bundle `visual.json`(供网页前端二次开发)+ `report.md`(中文摘要)。
## 命令
```bash
# 默认输出到 stdoutJSON
python scripts/research/visual.py --owner mindspore-Ecosystem --repo mindspore
# 输出可视化三件套到目录visual.html + visual.json + report.md
python scripts/research/visual.py --owner <OWNER> --repo <REPO> --weeks 26 --out ./out
# 可复现脚本(封装了上述流程)
bash skills/gitlink-research-visual/examples/research-visual-workflow.sh <OWNER> <REPO> [OUT_DIR] [WEEKS]
```
## 输出结构visual.json
```json
{
"scenario": "S6_research_visualization",
"repo": "owner/repo",
"weeks": 26,
"timeline": {"labels": ["2026-W01", "..."], "commits": [...], "issues": [...], "prs": [...]},
"heatmap": {"users": ["alice", "..."], "weeks": [...], "matrix": [[0,1,...], ...]},
"languages": {"Python": "99.7%"},
"milestones": [{"title": "v1.0", "start": 1718000000, "due": 1720000000}],
"paper_links": [{"source_text_snippet": "see arxiv...", "target": "https://arxiv.org/abs/2401.00012", "type": "arxiv"}],
"artifacts": [{"path": "data/train.csv", "category": "dataset", "name": "train.csv"}],
"artifact_summary": {"paper": 3, "dataset": 5, "model": 2, "benchmark": 1},
"meta": {"commit_count": 800, "issue_count": 20346, "pr_count": 9, "milestone_count": 4, "contributor_count": 6}
}
```
## 验证
已在真实科研仓库 **`mindspore-Ecosystem/mindspore`**default_branch=masterissue≈20346PR=9贡献者=6上验证
周分桶正确对齐 ISO 周(无 plotly 时优雅降级为 `visual.json` + `report.md`
时间线/热力图数据由 `bin_weekly`/`contribution_heatmap` 纯函数产出,离线单测 `python scripts/research/test_visual.py` 全部通过33 个用例,覆盖周分桶/热力矩阵/论文链接抽取/产物分类)。
## 兼容性
兼容 Claude Code 等 AI Agent本 SKILL.md 即为 Agent 编排依据,
Agent 可直接调上述命令并把产物HTML/JSON/Markdown读回做进一步解读与文案化。
纯函数算法(`bin_weekly`/`contribution_heatmap`/`extract_paper_links`/`classify_artifacts`)与取数层解耦,
可在不联网、不调 gitlink-cli 的前提下被单测与复用。

View File

@ -0,0 +1,48 @@
#!/usr/bin/env bash
# research-visual-workflow.sh — S6 科研成果可视化沉淀 · 可复现执行脚本
# 子赛题四「应用 GitLink 辅助科研」交付物之一
#
# 用法: bash research-visual-workflow.sh <OWNER> <REPO> [OUT_DIR] [WEEKS]
# 示例: bash research-visual-workflow.sh mindspore-Ecosystem mindspore ./out 26
set -euo pipefail
OWNER="${1:?用法: $0 <OWNER> <REPO> [OUT_DIR] [WEEKS]}"
REPO="${2:?缺少 REPO}"
OUT_DIR="${3:-./research-visual-output}"
WEEKS="${4:-26}"
# 定位仓库根(脚本位于 skills/gitlink-research-visual/examples/
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
REPO_ROOT="$(cd "$SCRIPT_DIR/../../.." && pwd)"
VISUAL="$REPO_ROOT/scripts/research/visual.py"
# 本地 Windows 开发可设 GITLINK_CLI=./gitlink-cli.exeLinux/容器走 PATH 默认值
: "${GITLINK_CLI:=gitlink-cli}"
echo "==> 目标仓库: $OWNER/$REPO"
echo "==> CLI: $GITLINK_CLI"
echo "==> 回溯周数: $WEEKS"
mkdir -p "$OUT_DIR"
GITLINK_CLI="$GITLINK_CLI" python "$VISUAL" \
--owner "$OWNER" --repo "$REPO" \
--weeks "$WEEKS" \
--out "$OUT_DIR"
echo
echo "==> 产物:"
ls -1 "$OUT_DIR"
echo
echo "==> 成果概览:"
python -c "
import json
d=json.load(open('$OUT_DIR/visual.json',encoding='utf-8'))
m=d['meta']; a=d['artifact_summary']
print(f\"仓库: {d['repo']} 回溯 {d['weeks']}\")
print(f\"活跃度: commits={m['commit_count']} issues={m['issue_count']} prs={m['pr_count']} 贡献者={m['contributor_count']}\")
print(f\"产物分类: paper={a['paper']} dataset={a['dataset']} model={a['model']} benchmark={a['benchmark']}\")
print(f\"抽取论文引用: {len(d['paper_links'])}\")
tl=d['timeline']
peak=max(tl['commits']) if tl['commits'] else 0
print(f\"开发节奏(最近窗口): 峰值 {peak} 提交/周\")
"