forked from Gitlink/gitlink-cli
Merge branch 'master' of https://gitlink.org.cn/whale_hihihi/gitlink-cli into surponess_br
# Conflicts: # .gitignore # gitlink-cli.exe
This commit is contained in:
commit
4e8b49444e
|
|
@ -20,7 +20,7 @@ workflow:
|
|||
ssh_ip: '"121.41.222.73"'
|
||||
ssh_port: '"22"'
|
||||
ssh_user: '"root"'
|
||||
ssh_cmd: '"if [ -d /root/gitlink-cli/.git ]; then cd /root/gitlink-cli && git fetch origin && git checkout master && git reset --hard origin/master; else rm -rf /root/gitlink-cli && git clone https://gitlink.org.cn/whale_hihihi/gitlink-cli.git /root/gitlink-cli && cd /root/gitlink-cli && git checkout master; fi && docker stop gitlink-cli 2>/dev/null; docker rm gitlink-cli 2>/dev/null; docker rmi gitlink-cli:latest 2>/dev/null; docker build --no-cache -t gitlink-cli:latest . && docker run -d --name gitlink-cli gitlink-cli:latest && echo Deploy success"'
|
||||
ssh_cmd: '"if [ -d /root/gitlink-cli/.git ]; then cd /root/gitlink-cli && git fetch origin && git checkout master && git reset --hard origin/master; else rm -rf /root/gitlink-cli && git clone https://gitlink.org.cn/whale_hihihi/gitlink-cli.git /root/gitlink-cli && cd /root/gitlink-cli && git checkout master; fi && docker stop gitlink-cli 2>/dev/null; docker rm gitlink-cli 2>/dev/null; docker rmi gitlink-cli:latest 2>/dev/null; docker build --no-cache -t gitlink-cli:latest . && docker run -d --name gitlink-cli -p 8080:8080 -e GITLINK_TOKEN=((gitlink_token.token)) gitlink-cli:latest && echo Deploy success"'
|
||||
needs:
|
||||
- start
|
||||
- ref: ssh_cmd_1
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
docs/
|
||||
doc/
|
||||
# Python(demo 网页后端)
|
||||
# Python(demo 网页后端)
|
||||
__pycache__/
|
||||
*.pyc
|
||||
demo/bin/
|
||||
37
Dockerfile
37
Dockerfile
|
|
@ -1,13 +1,36 @@
|
|||
# ============================================================
|
||||
# 多阶段构建:gitlink-cli 子赛题四网页终端
|
||||
# ============================================================
|
||||
# 阶段1 builder —— Go 静态编译
|
||||
# ============================================================
|
||||
FROM golang:1.26-alpine AS builder
|
||||
|
||||
ENV GOPROXY=https://goproxy.cn,direct
|
||||
WORKDIR /app
|
||||
WORKDIR /src
|
||||
|
||||
# 先拷依赖清单,利用 Docker 层缓存
|
||||
COPY go.mod go.sum ./
|
||||
RUN go mod download
|
||||
COPY . .
|
||||
RUN CGO_ENABLED=0 go build -ldflags="-s -w" -o gitlink-cli .
|
||||
|
||||
FROM alpine:latest
|
||||
RUN apk --no-cache add ca-certificates git
|
||||
COPY . .
|
||||
# modernc.org/sqlite 是 pure-Go,CGO_ENABLED=0 即可编译纯静态二进制
|
||||
RUN CGO_ENABLED=0 GOOS=linux go build -ldflags="-s -w" -o /out/gitlink-cli .
|
||||
|
||||
# ============================================================
|
||||
# 阶段2 runtime —— Python + Go 二进制
|
||||
# ============================================================
|
||||
FROM python:3.12-slim
|
||||
|
||||
# Go CLI 放入 PATH
|
||||
COPY --from=builder /out/gitlink-cli /usr/local/bin/gitlink-cli
|
||||
|
||||
# 科研算法层:先拷 requirements.txt 安装依赖(利用层缓存),再拷源码
|
||||
COPY scripts/research/requirements.txt /app/scripts/research/requirements.txt
|
||||
RUN pip install --no-cache-dir -r /app/scripts/research/requirements.txt
|
||||
COPY scripts/research/ /app/scripts/research/
|
||||
|
||||
WORKDIR /app
|
||||
COPY --from=builder /app/gitlink-cli /usr/local/bin/gitlink-cli
|
||||
ENTRYPOINT ["gitlink-cli"]
|
||||
|
||||
# 子赛题四网页终端 HTTP 服务
|
||||
EXPOSE 8080
|
||||
ENTRYPOINT ["gitlink-cli", "server", "--port", "8080", "--research-dir", "/app/scripts/research", "--work-dir", "/app/research-output"]
|
||||
|
|
|
|||
35
README.md
35
README.md
|
|
@ -104,7 +104,7 @@ The official [GitLink](https://www.gitlink.org.cn) CLI tool — built for humans
|
|||
| Category | Capabilities |
|
||||
|----------|-------------|
|
||||
| 📦 Repo | List, create, fork, delete repositories, view repo info, insights, and interactions |
|
||||
| 🐛 Issue | Create, update, close, batch close, comment on issues |
|
||||
| 🐛 Issue | Create, update, close, batch close/update/delete, comment on issues |
|
||||
| 🔖 Label | Create, list, update, delete issue labels |
|
||||
| 🔀 PR | Create, merge, review pull requests, view changed files |
|
||||
| 👥 Member | List, add, remove repository members, change roles, create and accept invite links |
|
||||
|
|
@ -114,6 +114,7 @@ The official [GitLink](https://www.gitlink.org.cn) CLI tool — built for humans
|
|||
| 🔧 CI | View builds, logs, CI/CD operations |
|
||||
| ⚙️ Pipeline | Run, inspect, enable, disable, delete pipeline workflows and logs |
|
||||
| 🔔 Webhook | Manage repo webhooks and test deliveries |
|
||||
| 📖 Wiki | List, view, create, update, and delete wiki pages |
|
||||
| 🔍 Search | Search repositories, users |
|
||||
| 📊 Dataset | Query research datasets by project |
|
||||
| 👤 User | View user profiles and info |
|
||||
|
|
@ -273,6 +274,28 @@ gitlink-cli webhook +test --owner Gitlink --repo forgeplus --id 68
|
|||
gitlink-cli webhook +tasks --owner Gitlink --repo forgeplus --id 68
|
||||
```
|
||||
|
||||
### Wiki Management
|
||||
|
||||
```bash
|
||||
# List wiki pages (table of contents)
|
||||
gitlink-cli wiki +list --owner Gitlink --repo forgeplus --project-id 12345
|
||||
|
||||
# View a wiki page by page name
|
||||
gitlink-cli wiki +view --owner Gitlink --repo forgeplus --project-id 12345 -n home
|
||||
|
||||
# Create a wiki page
|
||||
gitlink-cli wiki +create --owner Gitlink --repo forgeplus --project-id 12345 \
|
||||
-n getting-started -t "Getting Started" -c "# Getting Started Guide"
|
||||
|
||||
# Update a wiki page title and/or content
|
||||
gitlink-cli wiki +update --owner Gitlink --repo forgeplus --project-id 12345 -n home -t "New Title"
|
||||
gitlink-cli wiki +update --owner Gitlink --repo forgeplus --project-id 12345 -n home -c "# Updated content"
|
||||
gitlink-cli wiki +update --owner Gitlink --repo forgeplus --project-id 12345 -n home -t "New Title" -c "New content"
|
||||
|
||||
# Delete a wiki page
|
||||
gitlink-cli wiki +delete --owner Gitlink --repo forgeplus --project-id 12345 -n old-page
|
||||
```
|
||||
|
||||
### Member Management
|
||||
|
||||
```bash
|
||||
|
|
@ -322,6 +345,14 @@ gitlink-cli issue +batch-close --owner Gitlink --repo forgeplus --numbers 123,12
|
|||
# Batch close issues from a CSV file
|
||||
gitlink-cli issue +batch-close --owner Gitlink --repo forgeplus --from issues.csv
|
||||
|
||||
# Preview batch metadata update by API issue IDs
|
||||
# Note: --ids uses API issue IDs, not web URL issue numbers.
|
||||
gitlink-cli issue +batch-update --owner Gitlink --repo forgeplus --ids 101,102 --status-id 3 --priority-id 2 --dry-run
|
||||
|
||||
# Destructive batch delete requires both dry-run first and --yes for real execution
|
||||
gitlink-cli issue +batch-delete --owner Gitlink --repo forgeplus --ids 101,102 --dry-run
|
||||
gitlink-cli issue +batch-delete --owner Gitlink --repo forgeplus --ids 101,102 --yes
|
||||
|
||||
# Add a comment
|
||||
gitlink-cli issue +comment --owner Gitlink --repo forgeplus -i 123 -b "Fixed"
|
||||
|
||||
|
|
@ -707,7 +738,7 @@ See [skills/README.md](./skills/README.md) for details.
|
|||
|-------|-------------|
|
||||
| `gitlink-shared` | Authentication, global parameters, safety rules, API notes |
|
||||
| `gitlink-repo` | Repository operations (create, view, delete, fork, insights, etc.) |
|
||||
| `gitlink-issue` | Issue operations (create, update, close, comment, etc.) |
|
||||
| `gitlink-issue` | Issue operations (create, update, close, batch update/delete, comment, etc.) |
|
||||
| `gitlink-pr` | Pull request operations (create, merge, review, etc.) |
|
||||
| `gitlink-member` | Repository member and invite link management |
|
||||
| `gitlink-branch` | Branch management (create, delete, list, protect, unprotect) |
|
||||
|
|
|
|||
|
|
@ -104,7 +104,7 @@
|
|||
| 分类 | 能力 |
|
||||
|------|------|
|
||||
| 📦 仓库 | 列出、创建、Fork、删除仓库,查看仓库信息、洞察数据和互动状态 |
|
||||
| 🐛 Issue | 创建、更新、关闭、批量关闭、评论 Issue |
|
||||
| 🐛 Issue | 创建、更新、关闭、批量关闭/更新/删除、评论 Issue |
|
||||
| 🔖 标签 | 创建、列出、更新、删除 Issue 标签 |
|
||||
| 🔀 PR | 创建、合并、Review Pull Request,查看变更文件 |
|
||||
| 👥 成员 | 列出、添加、移除仓库成员,调整角色,生成和接受邀请链接 |
|
||||
|
|
@ -113,6 +113,7 @@
|
|||
| 🏢 组织 | 管理组织、成员、团队 |
|
||||
| 🔧 CI | 查看构建、日志、CI/CD 操作 |
|
||||
| ⚙️ Pipeline | 运行、查看、启停、删除流水线工作流并查询日志 |
|
||||
| 📖 Wiki | 列出、查看、创建、更新、删除 Wiki 页面 |
|
||||
| 🔍 搜索 | 搜索仓库、用户 |
|
||||
| 📊 数据集 | 按项目查询科研数据集 |
|
||||
| 👤 用户 | 查看用户资料和信息 |
|
||||
|
|
@ -284,6 +285,28 @@ gitlink-cli webhook +test --owner Gitlink --repo forgeplus --id 68
|
|||
gitlink-cli webhook +tasks --owner Gitlink --repo forgeplus --id 68
|
||||
```
|
||||
|
||||
### Wiki 管理
|
||||
|
||||
```bash
|
||||
# 列出 Wiki 页面(目录结构)
|
||||
gitlink-cli wiki +list --owner Gitlink --repo forgeplus --project-id 12345
|
||||
|
||||
# 查看 Wiki 页面
|
||||
gitlink-cli wiki +view --owner Gitlink --repo forgeplus --project-id 12345 -n home
|
||||
|
||||
# 创建 Wiki 页面
|
||||
gitlink-cli wiki +create --owner Gitlink --repo forgeplus --project-id 12345 \
|
||||
-n getting-started -t "快速开始" -c "# 快速开始指南"
|
||||
|
||||
# 更新 Wiki 页面标题和/或内容
|
||||
gitlink-cli wiki +update --owner Gitlink --repo forgeplus --project-id 12345 -n home -t "新标题"
|
||||
gitlink-cli wiki +update --owner Gitlink --repo forgeplus --project-id 12345 -n home -c "# 更新后的内容"
|
||||
gitlink-cli wiki +update --owner Gitlink --repo forgeplus --project-id 12345 -n home -t "新标题" -c "新内容"
|
||||
|
||||
# 删除 Wiki 页面
|
||||
gitlink-cli wiki +delete --owner Gitlink --repo forgeplus --project-id 12345 -n old-page
|
||||
```
|
||||
|
||||
### 成员管理
|
||||
|
||||
```bash
|
||||
|
|
@ -333,6 +356,14 @@ gitlink-cli issue +batch-close --owner Gitlink --repo forgeplus --numbers 123,12
|
|||
# 从 CSV 文件批量关闭 Issue
|
||||
gitlink-cli issue +batch-close --owner Gitlink --repo forgeplus --from issues.csv
|
||||
|
||||
# 按 API issue id 预览批量更新元数据
|
||||
# 注意:--ids 是 API issue id,不是网页 URL 中的 Issue 编号。
|
||||
gitlink-cli issue +batch-update --owner Gitlink --repo forgeplus --ids 101,102 --status-id 3 --priority-id 2 --dry-run
|
||||
|
||||
# 危险批量删除必须先 dry-run,真实执行还要显式 --yes
|
||||
gitlink-cli issue +batch-delete --owner Gitlink --repo forgeplus --ids 101,102 --dry-run
|
||||
gitlink-cli issue +batch-delete --owner Gitlink --repo forgeplus --ids 101,102 --yes
|
||||
|
||||
# 添加评论
|
||||
gitlink-cli issue +comment --owner Gitlink --repo forgeplus -i 123 -b "已修复"
|
||||
|
||||
|
|
@ -581,7 +612,7 @@ git push gitlink
|
|||
|-------|------|
|
||||
| `gitlink-shared` | 认证、全局参数、安全规则、API 注意事项 |
|
||||
| `gitlink-repo` | 仓库操作(创建、查看、删除、Fork、洞察数据等) |
|
||||
| `gitlink-issue` | Issue 操作(创建、更新、关闭、评论等) |
|
||||
| `gitlink-issue` | Issue 操作(创建、更新、关闭、批量更新/删除、评论等) |
|
||||
| `gitlink-pr` | Pull Request 操作(创建、合并、Review 等) |
|
||||
| `gitlink-member` | 仓库成员与邀请链接管理 |
|
||||
| `gitlink-release` | 发布管理(创建、编辑、更新、查看、删除等) |
|
||||
|
|
|
|||
|
|
@ -14,6 +14,7 @@ import (
|
|||
doctorCmd "github.com/gitlink-org/gitlink-cli/cmd/doctor"
|
||||
internalConfig "github.com/gitlink-org/gitlink-cli/internal/config"
|
||||
"github.com/gitlink-org/gitlink-cli/internal/i18n"
|
||||
serverCmd "github.com/gitlink-org/gitlink-cli/cmd/server"
|
||||
"github.com/gitlink-org/gitlink-cli/shortcuts"
|
||||
)
|
||||
|
||||
|
|
@ -58,6 +59,7 @@ func NewRootCmd(opts RootOptions, tr *i18n.Translator) (*cobra.Command, error) {
|
|||
rootCmd.AddCommand(apiCmd.NewAPICmd(tr))
|
||||
rootCmd.AddCommand(configCmd.NewConfigCmd(tr))
|
||||
rootCmd.AddCommand(doctorCmd.NewDoctorCmd(tr))
|
||||
rootCmd.AddCommand(serverCmd.NewServerCmd())
|
||||
rootCmd.AddCommand(newVersionCmd(version, tr))
|
||||
|
||||
shortcuts.RegisterAll(rootCmd, tr)
|
||||
|
|
|
|||
|
|
@ -0,0 +1,293 @@
|
|||
// Package server 提供子赛题四的「网页终端」演示服务(子赛题四·Phase 7)。
|
||||
//
|
||||
// 在云服务器上 `gitlink-cli server` 启动一个轻量 HTTP 服务:前端按 7 个科研场景按钮
|
||||
// 触发后端运行对应的 Python 算法脚本(scripts/research/*.py),并把 JSON / Mermaid /
|
||||
// 报告等产物回显/渲染。所有数据仍由 gitlink-cli 获取,本服务只做命令编排与产物转发。
|
||||
//
|
||||
// 安全:仅接受 7 个固定场景 + owner/repo/keyword 结构化参数(非任意 shell);
|
||||
// 若设置 --token / DEMO_TOKEN 环境变量,则请求需带 X-Demo-Token 头匹配。
|
||||
package server
|
||||
|
||||
import (
|
||||
"embed"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io/fs"
|
||||
"net/http"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/spf13/cobra"
|
||||
)
|
||||
|
||||
//go:embed static/*
|
||||
var staticFS embed.FS
|
||||
|
||||
const defaultPort = 8080
|
||||
|
||||
// 场景定义:前端按钮 ↔ 后端 Python 脚本。
|
||||
type scenarioDef struct {
|
||||
Key string `json:"key"`
|
||||
Label string `json:"label"`
|
||||
Desc string `json:"desc"`
|
||||
Script string `json:"script"`
|
||||
Needs []string `json:"needs"` // 需要的输入: "owner","repo","keyword"
|
||||
Timeout int `json:"timeout_s"`
|
||||
}
|
||||
|
||||
var scenarios = []scenarioDef{
|
||||
{Key: "s1", Label: "S1 仓库洞悉", Desc: "科研项目演进谱系 + 创新点", Script: "lineage.py", Needs: []string{"owner", "repo"}, Timeout: 180},
|
||||
{Key: "s2", Label: "S2 知识图谱", Desc: "科研领域知识图谱(networkx)", Script: "graph_build.py", Needs: []string{"keyword"}, Timeout: 240},
|
||||
{Key: "s3", Label: "S3 合规复现", Desc: "许可证/密钥/复现性检查", Script: "repro.py", Needs: []string{"owner", "repo"}, Timeout: 120},
|
||||
{Key: "s4", Label: "S4 协作匹配", Desc: "学者×缺口 智能匹配", Script: "match.py", Needs: []string{"owner", "repo"}, Timeout: 180},
|
||||
{Key: "s5", Label: "S5 进度预警", Desc: "周报 + 风险预警", Script: "report.py", Needs: []string{"owner", "repo"}, Timeout: 180},
|
||||
{Key: "s6", Label: "S6 成果可视化", Desc: "交互图表(plotly)", Script: "visual.py", Needs: []string{"owner", "repo"}, Timeout: 240},
|
||||
{Key: "hotspot", Label: "🔥 热点追踪", Desc: "科研热点全景观测:飙升项目+活跃讨论+主题热度+学者团队", Script: "hotspot.py", Needs: []string{"keyword"}, Timeout: 300},
|
||||
}
|
||||
|
||||
type Options struct {
|
||||
Port int
|
||||
ResearchDir string // scripts/research 目录
|
||||
WorkDir string // 产物输出根目录
|
||||
Token string // 可选鉴权 token
|
||||
}
|
||||
|
||||
func NewServerCmd() *cobra.Command {
|
||||
opts := Options{Port: defaultPort, ResearchDir: "scripts/research", WorkDir: "research-output"}
|
||||
cmd := &cobra.Command{
|
||||
Use: "server",
|
||||
Short: "启动子赛题四网页终端(HTTP 演示服务)",
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
return Run(opts)
|
||||
},
|
||||
}
|
||||
cmd.Flags().IntVarP(&opts.Port, "port", "p", defaultPort, "监听端口")
|
||||
cmd.Flags().StringVar(&opts.ResearchDir, "research-dir", "scripts/research", "scripts/research 目录")
|
||||
cmd.Flags().StringVar(&opts.WorkDir, "work-dir", "research-output", "产物输出根目录")
|
||||
cmd.Flags().StringVar(&opts.Token, "token", "", "可选鉴权 token(亦可用 DEMO_TOKEN 环境变量)")
|
||||
return cmd
|
||||
}
|
||||
|
||||
// Run 启动 HTTP 服务(阻塞)。
|
||||
func Run(opts Options) error {
|
||||
if t := os.Getenv("DEMO_TOKEN"); t != "" && opts.Token == "" {
|
||||
opts.Token = t
|
||||
}
|
||||
_ = os.MkdirAll(opts.WorkDir, 0o755)
|
||||
|
||||
mux := http.NewServeMux()
|
||||
h := &handler{opts: opts}
|
||||
|
||||
mux.HandleFunc("GET /api/scenarios", h.handleScenarios)
|
||||
mux.HandleFunc("POST /api/run", h.handleRun)
|
||||
mux.HandleFunc("GET /api/result/{key}", h.handleResult)
|
||||
mux.HandleFunc("GET /api/health", h.handleHealth)
|
||||
|
||||
sub, err := fs.Sub(staticFS, "static")
|
||||
if err != nil {
|
||||
return fmt.Errorf("static fs: %w", err)
|
||||
}
|
||||
mux.Handle("GET /", http.FileServer(http.FS(sub)))
|
||||
|
||||
addr := fmt.Sprintf(":%d", opts.Port)
|
||||
fmt.Fprintf(os.Stderr, "子赛题四 网页终端已启动: http://localhost%s\n", addr)
|
||||
fmt.Fprintf(os.Stderr, " research-dir=%s work-dir=%s auth=%v\n", opts.ResearchDir, opts.WorkDir, opts.Token != "")
|
||||
srv := &http.Server{Addr: addr, Handler: mux, ReadHeaderTimeout: 10 * time.Second}
|
||||
return srv.ListenAndServe()
|
||||
}
|
||||
|
||||
type handler struct {
|
||||
opts Options
|
||||
mu sync.Mutex // 串行化场景执行,避免并发打爆 GitLink API
|
||||
}
|
||||
|
||||
func (h *handler) authed(r *http.Request) bool {
|
||||
if h.opts.Token == "" {
|
||||
return true
|
||||
}
|
||||
return r.Header.Get("X-Demo-Token") == h.opts.Token
|
||||
}
|
||||
|
||||
func (h *handler) handleHealth(w http.ResponseWriter, r *http.Request) {
|
||||
writeJSON(w, map[string]any{"ok": true, "scenarios": len(scenarios)})
|
||||
}
|
||||
|
||||
func (h *handler) handleScenarios(w http.ResponseWriter, r *http.Request) {
|
||||
if !h.authed(r) {
|
||||
http.Error(w, "unauthorized", http.StatusUnauthorized)
|
||||
return
|
||||
}
|
||||
writeJSON(w, map[string]any{"ok": true, "scenarios": scenarios})
|
||||
}
|
||||
|
||||
// handleResult 返回某场景最近一次运行的产物(供 result.html 独立结果页按 key 读取,
|
||||
// URL 可刷新/分享,便于演示讲解)。无需鉴权串行锁——只读已落盘产物。
|
||||
func (h *handler) handleResult(w http.ResponseWriter, r *http.Request) {
|
||||
if !h.authed(r) {
|
||||
http.Error(w, "unauthorized", http.StatusUnauthorized)
|
||||
return
|
||||
}
|
||||
key := r.PathValue("key")
|
||||
sc, ok := findScenario(key)
|
||||
if !ok {
|
||||
writeJSON(w, map[string]any{"ok": false, "error": "unknown scenario: " + key})
|
||||
return
|
||||
}
|
||||
outDir := filepath.Join(h.opts.WorkDir, sc.Key)
|
||||
writeJSON(w, map[string]any{
|
||||
"ok": true,
|
||||
"scenario": sc.Key,
|
||||
"label": sc.Label,
|
||||
"desc": sc.Desc,
|
||||
"artifacts": readArtifacts(outDir),
|
||||
})
|
||||
}
|
||||
|
||||
type runRequest struct {
|
||||
Scenario string `json:"scenario"`
|
||||
Owner string `json:"owner"`
|
||||
Repo string `json:"repo"`
|
||||
Keyword string `json:"keyword"`
|
||||
}
|
||||
|
||||
func (h *handler) handleRun(w http.ResponseWriter, r *http.Request) {
|
||||
if !h.authed(r) {
|
||||
http.Error(w, "unauthorized", http.StatusUnauthorized)
|
||||
return
|
||||
}
|
||||
var req runRequest
|
||||
if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
|
||||
writeJSON(w, map[string]any{"ok": false, "error": "bad request: " + err.Error()})
|
||||
return
|
||||
}
|
||||
sc, ok := findScenario(req.Scenario)
|
||||
if !ok {
|
||||
writeJSON(w, map[string]any{"ok": false, "error": "unknown scenario: " + req.Scenario})
|
||||
return
|
||||
}
|
||||
for _, need := range sc.Needs {
|
||||
if (need == "keyword" && req.Keyword == "") ||
|
||||
(need == "owner" && req.Owner == "") ||
|
||||
(need == "repo" && req.Repo == "") {
|
||||
writeJSON(w, map[string]any{"ok": false, "error": "missing parameter: " + need})
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
// 串行执行:一次只跑一个场景,保护 GitLink API。
|
||||
h.mu.Lock()
|
||||
defer h.mu.Unlock()
|
||||
|
||||
scriptPath := filepath.Join(h.opts.ResearchDir, sc.Script)
|
||||
outDir := filepath.Join(h.opts.WorkDir, sc.Key)
|
||||
_ = os.MkdirAll(outDir, 0o755)
|
||||
|
||||
argv := []string{scriptPath, "--out", outDir}
|
||||
if contains(sc.Needs, "owner") {
|
||||
argv = append(argv, "--owner", req.Owner, "--repo", req.Repo)
|
||||
}
|
||||
if contains(sc.Needs, "keyword") {
|
||||
argv = append(argv, "--keywords", req.Keyword)
|
||||
}
|
||||
|
||||
// python3 优先,回退 python
|
||||
py, err := pythonBin()
|
||||
if err != nil {
|
||||
writeJSON(w, map[string]any{"ok": false, "error": err.Error()})
|
||||
return
|
||||
}
|
||||
cmd := exec.Command(py, argv...)
|
||||
cmd.Env = os.Environ()
|
||||
start := time.Now()
|
||||
out, err := cmd.CombinedOutput()
|
||||
dur := time.Since(start)
|
||||
resp := map[string]any{
|
||||
"ok": err == nil,
|
||||
"scenario": sc.Key,
|
||||
"command": py + " " + strings.Join(argv, " "),
|
||||
"duration": dur.Truncate(time.Millisecond).String(),
|
||||
"stdout": string(out),
|
||||
"out_dir": outDir,
|
||||
}
|
||||
if err != nil {
|
||||
resp["error"] = err.Error()
|
||||
}
|
||||
// 附带读取关键产物(json + 第一个 mmd + report.md),便于前端直接渲染
|
||||
resp["artifacts"] = readArtifacts(outDir)
|
||||
writeJSON(w, resp)
|
||||
}
|
||||
|
||||
func findScenario(key string) (scenarioDef, bool) {
|
||||
for _, s := range scenarios {
|
||||
if s.Key == key || strings.EqualFold(s.Key, key) {
|
||||
return s, true
|
||||
}
|
||||
}
|
||||
return scenarioDef{}, false
|
||||
}
|
||||
|
||||
func readArtifacts(dir string) map[string]string {
|
||||
out := map[string]string{}
|
||||
// json 产物(取第一个 *.json)
|
||||
if entries, err := os.ReadDir(dir); err == nil {
|
||||
for _, e := range entries {
|
||||
if e.IsDir() {
|
||||
continue
|
||||
}
|
||||
name := e.Name()
|
||||
switch {
|
||||
case strings.HasSuffix(name, ".json"):
|
||||
b, _ := os.ReadFile(filepath.Join(dir, name))
|
||||
out["json"] = string(b)
|
||||
case strings.HasSuffix(name, ".mmd"):
|
||||
b, _ := os.ReadFile(filepath.Join(dir, name))
|
||||
out["mermaid"] = string(b)
|
||||
case name == "visual.html":
|
||||
b, _ := os.ReadFile(filepath.Join(dir, name))
|
||||
out["html"] = string(b)
|
||||
case strings.HasSuffix(name, ".md"):
|
||||
// 第一份 .md 报告(report.md / weekly_report.md / compliance_report.md)
|
||||
if _, ok := out["report"]; !ok {
|
||||
b, _ := os.ReadFile(filepath.Join(dir, name))
|
||||
out["report"] = string(b)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func pythonBin() (string, error) {
|
||||
// 候选按 Linux 习惯 python3 优先,再 python / py(Windows)。
|
||||
// 必须实测能产出:Windows 的 WindowsApps\python3.exe 是 Store 桩,对 -c 也可能 exit 0 但不真正执行,
|
||||
// 故用「stdout 必须含 PYOK」来拦截桩。
|
||||
for _, name := range []string{"python3", "python", "py"} {
|
||||
path, err := exec.LookPath(name)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
if out, err := exec.Command(path, "-c", "print('PYOK')").Output(); err == nil &&
|
||||
strings.Contains(string(out), "PYOK") {
|
||||
return path, nil
|
||||
}
|
||||
}
|
||||
return "", fmt.Errorf("python 未安装;容器需内置 python3 并 pip install -r scripts/research/requirements.txt")
|
||||
}
|
||||
|
||||
func contains(xs []string, s string) bool {
|
||||
for _, x := range xs {
|
||||
if x == s {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func writeJSON(w http.ResponseWriter, v any) {
|
||||
w.Header().Set("Content-Type", "application/json; charset=utf-8")
|
||||
_ = json.NewEncoder(w).Encode(v)
|
||||
}
|
||||
|
|
@ -0,0 +1,485 @@
|
|||
<!DOCTYPE html>
|
||||
<html lang="zh-CN">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
||||
<title>科研热点追踪 · GitLink Research Atlas</title>
|
||||
<link rel="preconnect" href="https://fonts.googleapis.com">
|
||||
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
|
||||
<link href="https://fonts.googleapis.com/css2?family=Cormorant+Garamond:wght@500;600;700&family=Inter:wght@400;500;600;700&family=JetBrains+Mono:wght@500;700&family=Noto+Serif+SC:wght@600;700&family=Noto+Sans+SC:wght@400;500;700&display=swap" rel="stylesheet">
|
||||
<style>
|
||||
:root{
|
||||
--bg:#F7F9FC; --card:#FFFFFF; --ink:#1E2A44; --ink2:#475569; --mute:#94A3B8;
|
||||
--line:#E6EBF2; --indigo:#4F6BED; --coral:#F26B5E; --mint:#2EC4B6;
|
||||
--t-blue:#E8EEFF; --t-coral:#FFEDEA; --t-mint:#E6F7F4; --t-amber:#FEF3C7;
|
||||
--serif:"Cormorant Garamond","Noto Serif SC",Georgia,serif;
|
||||
--sans:"Inter","Noto Sans SC",-apple-system,"PingFang SC","Microsoft YaHei",sans-serif;
|
||||
--mono:"JetBrains Mono",ui-monospace,SFMono-Regular,Menlo,Consolas,monospace;
|
||||
--shadow:0 1px 2px rgba(30,42,68,.04),0 6px 18px rgba(30,42,68,.06);
|
||||
--shadow-sm:0 1px 2px rgba(30,42,68,.06);
|
||||
}
|
||||
*{box-sizing:border-box;margin:0;padding:0}
|
||||
html,body{min-height:100%}
|
||||
body{font-family:var(--sans);color:var(--ink);background:var(--bg);
|
||||
background-image:radial-gradient(1200px 480px at 80% -8%,#EAF0FF 0%,rgba(234,240,255,0) 60%),
|
||||
radial-gradient(900px 420px at 0% 0%,#FFF1EE 0%,rgba(255,241,238,0) 55%);
|
||||
-webkit-font-smoothing:antialiased}
|
||||
.wrap{max-width:1180px;margin:0 auto;padding:28px 36px 72px}
|
||||
|
||||
/* header */
|
||||
.hdr{display:flex;align-items:center;gap:14px;margin-bottom:22px}
|
||||
.back{display:inline-flex;align-items:center;gap:6px;font-size:13px;font-weight:600;color:var(--indigo);
|
||||
text-decoration:none;padding:7px 14px;border-radius:10px;background:var(--t-blue);white-space:nowrap}
|
||||
.back:hover{background:#D6DEFF}
|
||||
.hdr .titles h1{font-family:var(--serif);font-weight:700;font-size:30px;line-height:1.15}
|
||||
.hdr .titles .sub{font-size:13px;color:var(--mute);margin-top:3px}
|
||||
|
||||
/* control panel */
|
||||
.ctl{background:var(--card);border:1px solid var(--line);border-radius:16px;box-shadow:var(--shadow);
|
||||
padding:22px 24px;margin-bottom:22px}
|
||||
.row{display:flex;align-items:center;gap:12px;flex-wrap:wrap}
|
||||
.row + .row{margin-top:14px}
|
||||
.row .lbl{font-size:12px;font-weight:600;color:var(--ink2);min-width:84px;letter-spacing:.02em}
|
||||
.search-input{flex:1;min-width:260px;display:flex;align-items:center;gap:8px;background:#F8FAFD;
|
||||
border:1px solid var(--line);border-radius:11px;padding:11px 14px}
|
||||
.search-input:focus-within{border-color:var(--indigo);background:#fff;box-shadow:0 0 0 3px rgba(79,107,237,.12)}
|
||||
.search-input input{border:0;outline:0;font-family:var(--mono);font-size:14px;width:100%;background:transparent;color:var(--ink)}
|
||||
.ex-chips{display:flex;gap:7px;flex-wrap:wrap}
|
||||
.ex-chips .ex{font-size:12px;color:var(--ink2);background:#F1F5F9;border:1px solid var(--line);
|
||||
padding:5px 11px;border-radius:999px;cursor:pointer;transition:.15s}
|
||||
.ex-chips .ex:hover{background:var(--t-blue);border-color:#C7D2FE;color:var(--indigo)}
|
||||
.angles{display:flex;gap:9px;flex-wrap:wrap}
|
||||
.ang{display:inline-flex;align-items:center;gap:7px;font-size:13px;font-weight:500;padding:8px 14px;
|
||||
border-radius:11px;border:1px solid var(--line);background:#fff;cursor:pointer;transition:.15s;user-select:none}
|
||||
.ang .dot{width:9px;height:9px;border-radius:50%;background:var(--mute);transition:.15s}
|
||||
.ang.on{border-color:var(--indigo);background:#F5F8FF;color:var(--ink)}
|
||||
.ang.on .dot{background:var(--indigo);box-shadow:0 0 0 3px rgba(79,107,237,.18)}
|
||||
.ang[data-a="trending"] .dot{background:var(--coral)} .ang.on[data-a="trending"] .dot{background:var(--coral);box-shadow:0 0 0 3px rgba(242,107,94,.2)}
|
||||
.ang[data-a="active"] .dot{background:#F59E0B} .ang.on[data-a="active"] .dot{background:#F59E0B;box-shadow:0 0 0 3px rgba(245,158,11,.2)}
|
||||
.ang[data-a="topics"] .dot{background:#06B6D4} .ang.on[data-a="topics"] .dot{background:#06B6D4;box-shadow:0 0 0 3px rgba(6,182,212,.2)}
|
||||
.ang[data-a="scholars"] .dot{background:var(--mint)} .ang.on[data-a="scholars"] .dot{background:var(--mint);box-shadow:0 0 0 3px rgba(46,196,182,.2)}
|
||||
select.lim{font-family:var(--mono);font-size:13px;padding:8px 10px;border:1px solid var(--line);border-radius:10px;background:#fff;color:var(--ink)}
|
||||
.btn-run{margin-left:auto;font-size:14px;font-weight:600;color:#fff;background:linear-gradient(135deg,var(--indigo),#7C8CF5);
|
||||
border:0;border-radius:11px;padding:11px 22px;cursor:pointer;box-shadow:var(--shadow-sm);transition:.15s}
|
||||
.btn-run:hover{filter:brightness(1.05);transform:translateY(-1px)}
|
||||
.btn-run:disabled{opacity:.6;cursor:not-allowed;transform:none}
|
||||
.status{font-family:var(--mono);font-size:12px;color:var(--mute);margin-top:14px;display:flex;align-items:center;gap:8px}
|
||||
.status .ok{color:var(--mint)} .status .err{color:var(--coral)}
|
||||
.spin{width:14px;height:14px;border-radius:50%;border:2px solid var(--t-blue);border-top-color:var(--indigo);animation:rot .8s linear infinite}
|
||||
@keyframes rot{to{transform:rotate(360deg)}}
|
||||
|
||||
/* summary banner */
|
||||
.summary{display:grid;grid-template-columns:repeat(4,1fr);gap:12px;margin-bottom:18px}
|
||||
.sum-tile{border:1px solid var(--line);border-radius:13px;padding:14px 16px;background:var(--card);box-shadow:var(--shadow-sm)}
|
||||
.sum-tile .sl{font-size:11px;color:var(--mute);font-weight:600;letter-spacing:.02em}
|
||||
.sum-tile .sv{font-family:var(--mono);font-weight:700;font-size:26px;line-height:1.1;margin-top:4px;color:var(--ink)}
|
||||
.sum-tile .ss{font-size:11px;color:var(--mute);margin-top:2px}
|
||||
|
||||
/* sections */
|
||||
.sec{background:var(--card);border:1px solid var(--line);border-radius:16px;box-shadow:var(--shadow);
|
||||
padding:20px 24px;margin-top:18px}
|
||||
.sec-head{display:flex;align-items:center;gap:10px;margin-bottom:16px}
|
||||
.sec-head .ico{width:30px;height:30px;border-radius:9px;display:grid;place-items:center;font-size:15px}
|
||||
.sec-head h2{font-family:var(--serif);font-weight:700;font-size:21px}
|
||||
.sec-head .eyebrow{font-family:var(--mono);font-size:10px;letter-spacing:.16em;text-transform:uppercase;color:var(--mute);display:block;margin-top:1px}
|
||||
.sec-head .cnt{font-family:var(--mono);font-size:12px;color:var(--mute);margin-left:auto}
|
||||
.ic-coral{background:var(--t-coral);color:var(--coral)} .ic-amber{background:var(--t-amber);color:#B45309}
|
||||
.ic-cyan{background:#E0F7FA;color:#0891B2} .ic-mint{background:var(--t-mint);color:#0E7E72}
|
||||
|
||||
/* trending table */
|
||||
.tbl{width:100%;border-collapse:collapse;font-size:13px}
|
||||
.tbl th{text-align:left;font-size:11px;color:var(--mute);font-weight:600;padding:8px 10px;border-bottom:1px solid var(--line);text-transform:uppercase;letter-spacing:.04em}
|
||||
.tbl td{padding:10px;border-bottom:1px solid var(--line);vertical-align:top}
|
||||
.tbl tr:hover td{background:#F8FAFD}
|
||||
.tbl .repo{font-family:var(--mono);font-weight:600;color:var(--indigo)}
|
||||
.tbl .num{font-family:var(--mono);font-weight:700;text-align:right}
|
||||
.lang-pill{font-size:11px;padding:2px 8px;border-radius:6px;background:var(--t-blue);color:var(--indigo)}
|
||||
.scorebar{height:6px;border-radius:4px;background:var(--line);overflow:hidden;margin-top:4px;min-width:70px}
|
||||
.scorebar i{display:block;height:100%;background:linear-gradient(90deg,var(--coral),#FF9B8F)}
|
||||
.kw-chip{display:inline-block;font-size:10px;font-weight:600;padding:2px 7px;border-radius:5px;background:var(--t-amber);color:#92400E;margin-right:3px}
|
||||
|
||||
/* active discussions */
|
||||
.disc{display:flex;flex-direction:column;gap:10px}
|
||||
.disc-item{display:flex;gap:12px;align-items:flex-start;padding:12px 14px;border:1px solid var(--line);border-radius:11px;background:#FCFDFE}
|
||||
.disc-item .badge{font-size:10px;font-weight:700;padding:3px 8px;border-radius:6px;text-transform:uppercase;letter-spacing:.05em;flex-shrink:0;margin-top:1px}
|
||||
.badge.issue{background:var(--t-blue);color:var(--indigo)} .badge.pr{background:var(--t-mint);color:#0E7E72}
|
||||
.disc-item .main{flex:1;min-width:0}
|
||||
.disc-item .ttl{font-size:13px;font-weight:500;line-height:1.4}
|
||||
.disc-item .meta{font-size:11px;color:var(--mute);margin-top:3px;font-family:var(--mono)}
|
||||
.disc-item .cmt{font-family:var(--mono);font-weight:700;color:#B45309;font-size:13px;flex-shrink:0}
|
||||
|
||||
/* topic bars */
|
||||
.tlist{display:flex;flex-direction:column;gap:11px}
|
||||
.trow{display:flex;align-items:center;gap:12px}
|
||||
.trow .tname{width:180px;font-size:13px;font-weight:500;font-family:var(--mono);color:var(--ink)}
|
||||
.trow .tbar{flex:1;height:24px;background:#F1F5F9;border-radius:7px;overflow:hidden}
|
||||
.trow .tbar i{display:block;height:100%;background:linear-gradient(90deg,#06B6D4,#22D3EE);border-radius:7px}
|
||||
.trow .tval{font-family:var(--mono);font-weight:700;font-size:13px;width:48px;text-align:right}
|
||||
|
||||
/* scholars grid */
|
||||
.sgrid{display:grid;grid-template-columns:repeat(auto-fill,minmax(200px,1fr));gap:12px}
|
||||
.scard{border:1px solid var(--line);border-radius:11px;padding:13px 15px;background:#FCFDFE}
|
||||
.scard .nm{font-family:var(--mono);font-weight:600;color:var(--ink);font-size:13px}
|
||||
.scard .ct{font-size:12px;color:var(--mute);margin-top:3px}
|
||||
.scard.org{border-left:3px solid var(--indigo)}
|
||||
|
||||
/* repos network mini-list */
|
||||
.rlist{display:flex;flex-direction:column;gap:6px}
|
||||
.rlist .rli{font-family:var(--mono);font-size:11px;color:var(--mute);padding:2px 0}
|
||||
|
||||
/* empty / loading / placeholder */
|
||||
.empty{padding:50px 20px;text-align:center;color:var(--mute);font-size:14px}
|
||||
.placeholder{margin-top:18px;padding:60px 20px;text-align:center;color:var(--mute);background:var(--card);
|
||||
border:1px dashed var(--line);border-radius:16px;font-size:14px}
|
||||
.placeholder b{color:var(--ink2)}
|
||||
|
||||
/* skeleton loading */
|
||||
@keyframes shimmer{0%{background-position:-400px 0}100%{background-position:400px 0}}
|
||||
.skel{background:linear-gradient(90deg,#F1F5F9 25%,#E6EBF2 50%,#F1F5F9 75%);background-size:800px 100%;animation:shimmer 1.5s infinite;border-radius:8px}
|
||||
|
||||
/* responsive */
|
||||
@media(max-width:760px){
|
||||
.wrap{padding:16px}
|
||||
.summary{grid-template-columns:repeat(2,1fr)}
|
||||
.sgrid{grid-template-columns:repeat(auto-fill,minmax(150px,1fr))}
|
||||
.row .lbl{min-width:60px;font-size:11px}
|
||||
}
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<div class="wrap">
|
||||
<div class="hdr">
|
||||
<a class="back" href="index.html">← 仪表盘</a>
|
||||
<div class="titles">
|
||||
<h1>科研热点追踪</h1>
|
||||
<div class="sub">输入领域/关键词,勾选追踪角度,扫描 GitLink 科研仓库 → 飙升项目 · 活跃讨论 · 主题热度 · 学者团队</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- control panel -->
|
||||
<div class="ctl">
|
||||
<div class="row">
|
||||
<span class="lbl">领域 / 关键词</span>
|
||||
<label class="search-input"><span>🔍</span>
|
||||
<input id="keyword" value="deep learning,机器学习" placeholder='输入研究领域或技术关键词,逗号分隔多个(如 "计算机视觉,NLP,transformer")'>
|
||||
</label>
|
||||
<button class="btn-run" id="runBtn">开始追踪 ▶</button>
|
||||
</div>
|
||||
<div class="row">
|
||||
<span class="lbl">示例</span>
|
||||
<div class="ex-chips" id="exChips">
|
||||
<span class="ex">深度学习</span><span class="ex">computer vision</span>
|
||||
<span class="ex">自然语言处理</span><span class="ex">强化学习</span>
|
||||
<span class="ex">knowledge graph</span><span class="ex">联邦学习</span>
|
||||
<span class="ex">LLM agent</span><span class="ex">自动驾驶</span>
|
||||
</div>
|
||||
</div>
|
||||
<div class="row">
|
||||
<span class="lbl">追踪角度</span>
|
||||
<div class="angles" id="angles">
|
||||
<span class="ang on" data-a="trending"><span class="dot"></span>🔥 飙升项目</span>
|
||||
<span class="ang on" data-a="active"><span class="dot"></span>💬 活跃讨论</span>
|
||||
<span class="ang on" data-a="topics"><span class="dot"></span>📊 热门主题</span>
|
||||
<span class="ang on" data-a="scholars"><span class="dot"></span>👥 核心学者·团队</span>
|
||||
</div>
|
||||
<span class="lbl" style="margin-left:auto">仓库上限</span>
|
||||
<select class="lim" id="lim">
|
||||
<option>5</option><option>8</option><option selected>12</option><option>20</option><option>30</option>
|
||||
</select>
|
||||
</div>
|
||||
<div class="status" id="status"><span>就绪 · 点"开始追踪"扫描 GitLink</span></div>
|
||||
</div>
|
||||
|
||||
<!-- summary tiles (hidden until first run) -->
|
||||
<div class="summary" id="summary" style="display:none"></div>
|
||||
|
||||
<!-- results -->
|
||||
<div id="results">
|
||||
<div class="placeholder">输入关键词并选择角度,点 <b>开始追踪</b> —— 服务器将通过 gitlink-cli 实时搜索并构建该领域的科研热点全景。</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<script>
|
||||
/* ========== 工具函数 ========== */
|
||||
function esc(s){return String(s||"").replace(/[&<>]/g,c=>({'&':'&','<':'<','>':'>'}[c]));}
|
||||
function token(){return localStorage.getItem("demo_token")||"";}
|
||||
|
||||
/* ========== angle 切换 + 示例芯片 ========== */
|
||||
document.querySelectorAll(".ang").forEach(a=>a.onclick=()=>a.classList.toggle("on"));
|
||||
document.querySelectorAll(".ex").forEach(e=>e.onclick=()=>{
|
||||
document.getElementById("keyword").value=e.textContent.trim();
|
||||
});
|
||||
|
||||
/* ========== 状态栏 ========== */
|
||||
function setStatus(html,cls){
|
||||
const s=document.getElementById("status");
|
||||
s.innerHTML=html;
|
||||
s.className="status"+(cls?" "+cls:"");
|
||||
}
|
||||
|
||||
/* ========== 主运行 ========== */
|
||||
async function run(){
|
||||
const kw=document.getElementById("keyword").value.trim();
|
||||
if(!kw){setStatus("请输入关键词","err");return;}
|
||||
|
||||
const angles=[...document.querySelectorAll(".ang.on")].map(a=>a.dataset.a);
|
||||
if(!angles.length){setStatus("请至少选择一个追踪角度","err");return;}
|
||||
|
||||
const btn=document.getElementById("runBtn");
|
||||
btn.disabled=true;
|
||||
|
||||
setStatus('<span class="spin"></span> 扫描中:搜索 "'+esc(kw)+'" → 拉取仓库详情 → 批量获取 Issue/PR 讨论…(约 30–90s,取决于仓库数和 API 响应速度)');
|
||||
document.getElementById("results").innerHTML="";
|
||||
document.getElementById("summary").style.display="none";
|
||||
|
||||
try{
|
||||
const r=await fetch("/api/run",{
|
||||
method:"POST",
|
||||
headers:{"Content-Type":"application/json","X-Demo-Token":token()},
|
||||
body:JSON.stringify({scenario:"hotspot",keyword:kw})
|
||||
});
|
||||
const j=await r.json();
|
||||
|
||||
if(!j.ok){
|
||||
const errMsg=j.error||j.stdout||"未知错误";
|
||||
setStatus("✗ 运行失败: "+esc(errMsg)+" · "+(j.duration||""),"err");
|
||||
btn.disabled=false;
|
||||
return;
|
||||
}
|
||||
|
||||
// 解析 JSON
|
||||
let obj=null;
|
||||
try{
|
||||
const raw=(j.artifacts&&j.artifacts.json)||(j.stdout||"");
|
||||
obj=typeof raw==="string"?JSON.parse(raw):raw;
|
||||
}catch(e){
|
||||
setStatus("✗ 解析 JSON 产物失败 · 可能是 Python 脚本异常: "+(e.message||""),"err");
|
||||
btn.disabled=false;
|
||||
return;
|
||||
}
|
||||
|
||||
if(!obj||typeof obj!=="object"){
|
||||
setStatus("✗ 返回数据为空 · 请确认目标关键词在 GitLink 上有结果","err");
|
||||
btn.disabled=false;
|
||||
return;
|
||||
}
|
||||
|
||||
const meta=obj.meta||{};
|
||||
setStatus(
|
||||
"✓ 完成 · "+(j.duration||"")+
|
||||
" · 仓库 "+meta.repo_count+" 个"+
|
||||
" · 讨论 "+meta.discussion_count+" 条"+
|
||||
" · 主题 "+meta.topic_count+" 个"+
|
||||
" · 学者 "+meta.scholar_count+" 位",
|
||||
"ok"
|
||||
);
|
||||
|
||||
// 汇总卡片
|
||||
renderSummary(meta);
|
||||
// 结果
|
||||
render(angles,obj);
|
||||
|
||||
}catch(e){
|
||||
setStatus("✗ 请求失败: "+esc(e.message||e),"err");
|
||||
}finally{
|
||||
btn.disabled=false;
|
||||
}
|
||||
}
|
||||
|
||||
/* ========== 汇总卡片 ========== */
|
||||
function renderSummary(meta){
|
||||
const el=document.getElementById("summary");
|
||||
el.style.display="grid";
|
||||
el.innerHTML=
|
||||
tile("indigo","扫描仓库",meta.repo_count||0,"个")+
|
||||
tile("coral","活跃讨论",meta.discussion_count||0,"条")+
|
||||
tile("cyan","热门主题",meta.topic_count||0,"个")+
|
||||
tile("mint","核心学者",meta.scholar_count||0,"位");
|
||||
}
|
||||
function tile(c,label,val,unit){
|
||||
const colors={indigo:"var(--indigo)",coral:"var(--coral)",cyan:"#06B6D4",mint:"var(--mint)"};
|
||||
const bg={indigo:"var(--t-blue)",coral:"var(--t-coral)",cyan:"#E0F7FA",mint:"var(--t-mint)"};
|
||||
return `<div class="sum-tile" style="border-top:3px solid ${colors[c]||colors.indigo}">
|
||||
<div class="sl">${label}</div><div class="sv">${val}</div><div class="ss">${unit}</div></div>`;
|
||||
}
|
||||
|
||||
/* ========== 渲染各个角度 ========== */
|
||||
function render(angles,o){
|
||||
const box=document.getElementById("results");
|
||||
let h="";
|
||||
const maxScore=Math.max(1,...(o.trending_repos||[]).map(r=>r.score||0));
|
||||
|
||||
// --- 🔥 飙升项目 ---
|
||||
if(angles.includes("trending")){
|
||||
const tr=o.trending_repos||[];
|
||||
h+=`<div class="sec">
|
||||
<div class="sec-head">
|
||||
<div class="ico ic-coral">🔥</div>
|
||||
<div>
|
||||
<h2>飙升 / 热门项目</h2>
|
||||
<span class="eyebrow">Trending repos · 热度 = Star + Fork×2 + 近期更新加成</span>
|
||||
</div>
|
||||
<span class="cnt">${tr.length} 个</span>
|
||||
</div>`;
|
||||
if(tr.length){
|
||||
h+=`<table class="tbl"><thead><tr>
|
||||
<th>仓库</th><th>语言</th><th class="num">★ Star</th><th class="num">⑂ Fork</th><th class="num">日均★</th><th>热度</th><th>更新</th></tr></thead><tbody>`;
|
||||
tr.forEach(r=>{
|
||||
const kws=(r.matched_keywords||[]).map(k=>`<span class="kw-chip">${esc(k)}</span>`).join("");
|
||||
h+=`<tr>
|
||||
<td><div class="repo">${esc(r.repo)}</div>
|
||||
${kws?`<div style="margin-top:3px">${kws}</div>`:""}
|
||||
<div style="font-size:11px;color:var(--mute);margin-top:2px">${esc(r.description||"").slice(0,100)}</div></td>
|
||||
<td>${r.language?`<span class="lang-pill">${esc(r.language)}</span>`:"—"}</td>
|
||||
<td class="num">${r.stars}</td>
|
||||
<td class="num">${r.forks}</td>
|
||||
<td class="num">${r.velocity}</td>
|
||||
<td>
|
||||
<div class="scorebar"><i style="width:${Math.round(r.score/maxScore*100)}%"></i></div>
|
||||
<div style="font-size:11px;color:var(--mute);margin-top:2px;font-family:var(--mono)">${r.score}</div>
|
||||
</td>
|
||||
<td style="font-family:var(--mono);font-size:12px;color:var(--ink2)">${esc(r.updated||"—")}</td>
|
||||
</tr>`;
|
||||
});
|
||||
h+=`</tbody></table>`;
|
||||
}else{
|
||||
h+=`<div class="empty">未找到匹配仓库</div>`;
|
||||
}
|
||||
h+=`</div>`;
|
||||
}
|
||||
|
||||
// --- 💬 活跃讨论 ---
|
||||
if(angles.includes("active")){
|
||||
const ac=o.active_discussions||[];
|
||||
h+=`<div class="sec">
|
||||
<div class="sec-head">
|
||||
<div class="ico ic-amber">💬</div>
|
||||
<div>
|
||||
<h2>活跃讨论</h2>
|
||||
<span class="eyebrow">Active issues / PRs · 按评论数排序 · 来自所有扫描仓库</span>
|
||||
</div>
|
||||
<span class="cnt">${ac.length} 条</span>
|
||||
</div>`;
|
||||
if(ac.length){
|
||||
h+=`<div class="disc">`;
|
||||
ac.slice(0,30).forEach(a=>{
|
||||
h+=`<div class="disc-item">
|
||||
<span class="badge ${a.type}">${a.type==="pr"?"PR":"Issue"}</span>
|
||||
<div class="main">
|
||||
<div class="ttl">${esc(a.title||"(无标题)")}</div>
|
||||
<div class="meta">${esc(a.repo)} · #${esc(a.number||"")} · ${esc(a.state||"")}</div>
|
||||
</div>
|
||||
<span class="cmt">${a.comments} 💬</span>
|
||||
</div>`;
|
||||
});
|
||||
h+=`</div>`;
|
||||
}else{
|
||||
h+=`<div class="empty">该领域暂无活跃讨论(所有仓库的开放 Issue/PR 评论数均为 0)</div>`;
|
||||
}
|
||||
h+=`</div>`;
|
||||
}
|
||||
|
||||
// --- 📊 热门主题 ---
|
||||
if(angles.includes("topics")){
|
||||
const tp=o.topic_heat||[];
|
||||
const maxT=Math.max(1,...tp.map(t=>t.count));
|
||||
h+=`<div class="sec">
|
||||
<div class="sec-head">
|
||||
<div class="ico ic-cyan">📊</div>
|
||||
<div>
|
||||
<h2>热门主题</h2>
|
||||
<span class="eyebrow">Topic heat · 在仓库描述/README 中命中的主题词频</span>
|
||||
</div>
|
||||
<span class="cnt">${tp.length} 个</span>
|
||||
</div>`;
|
||||
if(tp.length){
|
||||
h+=`<div class="tlist">`;
|
||||
tp.forEach(t=>{
|
||||
h+=`<div class="trow">
|
||||
<span class="tname">${esc(t.topic)}</span>
|
||||
<div class="tbar"><i style="width:${Math.round(t.count/maxT*100)}%"></i></div>
|
||||
<span class="tval">${t.count}</span>
|
||||
</div>`;
|
||||
});
|
||||
h+=`</div>`;
|
||||
}else{
|
||||
h+=`<div class="empty">未识别到明确主题(仓库描述/README 不包含已知科研关键词)</div>`;
|
||||
}
|
||||
h+=`</div>`;
|
||||
}
|
||||
|
||||
// --- 👥 核心学者·团队 ---
|
||||
if(angles.includes("scholars")){
|
||||
const sc=o.core_scholars||[];
|
||||
const tm=o.core_teams||[];
|
||||
h+=`<div class="sec">
|
||||
<div class="sec-head">
|
||||
<div class="ico ic-mint">👥</div>
|
||||
<div>
|
||||
<h2>核心学者 · 团队</h2>
|
||||
<span class="eyebrow">Key scholars & teams · 按关联仓库数排列</span>
|
||||
</div>
|
||||
<span class="cnt">学者 ${(sc||[]).length} · 团队 ${(tm||[]).length}</span>
|
||||
</div>`;
|
||||
|
||||
// 学者
|
||||
if(sc.length){
|
||||
h+=`<div style="margin-bottom:14px"><div style="font-size:12px;font-weight:600;color:var(--ink2);margin-bottom:8px">🧑 学者</div>`;
|
||||
h+=`<div class="sgrid">`;
|
||||
sc.slice(0,12).forEach(s=>{
|
||||
const repoNames=(s.repos||[]).slice(0,5);
|
||||
h+=`<div class="scard">
|
||||
<div class="nm">${esc(s.login)}</div>
|
||||
<div class="ct">关联 ${s.repo_count} 个仓库</div>
|
||||
${repoNames.length?`<div class="rlist">${repoNames.map(rn=>`<span class="rli">${esc(rn)}</span>`).join("")}</div>`:""}
|
||||
</div>`;
|
||||
});
|
||||
h+=`</div></div>`;
|
||||
}
|
||||
|
||||
// 团队
|
||||
if(tm.length){
|
||||
h+=`<div><div style="font-size:12px;font-weight:600;color:var(--ink2);margin-bottom:8px">🏛 活跃组织/团队</div>`;
|
||||
h+=`<div class="sgrid">`;
|
||||
tm.slice(0,8).forEach(t=>{
|
||||
h+=`<div class="scard org">
|
||||
<div class="nm">${esc(t.login)}</div>
|
||||
<div class="ct">${t.repo_count} 个仓库</div>
|
||||
</div>`;
|
||||
});
|
||||
h+=`</div></div>`;
|
||||
}
|
||||
|
||||
if(!sc.length&&!tm.length){
|
||||
h+=`<div class="empty">未识别到学者/团队(仓库无贡献者数据)</div>`;
|
||||
}
|
||||
h+=`</div>`;
|
||||
}
|
||||
|
||||
box.innerHTML=h||'<div class="empty" style="padding:70px 20px">无匹配角度的数据 —— 请选择至少一个追踪角度</div>';
|
||||
}
|
||||
|
||||
/* ========== 事件绑定 ========== */
|
||||
document.getElementById("runBtn").onclick=run;
|
||||
document.getElementById("keyword").addEventListener("keydown",e=>{if(e.key==="Enter")run();});
|
||||
|
||||
/* ========== 页面加载时尝试载入上次的 hotspot 结果 ========== */
|
||||
(async function init(){
|
||||
try{
|
||||
const r=await fetch("/api/result/hotspot",{headers:{"X-Demo-Token":token()}});
|
||||
const j=await r.json();
|
||||
if(j.ok&&j.artifacts&&j.artifacts.json){
|
||||
const obj=JSON.parse(j.artifacts.json);
|
||||
if(obj&&obj.trending_repos){
|
||||
const angles=[...document.querySelectorAll(".ang.on")].map(a=>a.dataset.a);
|
||||
const meta=obj.meta||{};
|
||||
renderSummary(meta);
|
||||
render(angles,obj);
|
||||
setStatus("✓ 已载入上次扫描 · 点"开始追踪"用新关键词重新扫描","ok");
|
||||
}
|
||||
}
|
||||
}catch(e){}
|
||||
})();
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
|
|
@ -0,0 +1,594 @@
|
|||
<!DOCTYPE html>
|
||||
<html lang="zh-CN">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
||||
<title>GitLink Research Atlas · 科研代码图谱智能体</title>
|
||||
<link rel="preconnect" href="https://fonts.googleapis.com">
|
||||
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
|
||||
<link href="https://fonts.googleapis.com/css2?family=Cormorant+Garamond:wght@500;600;700&family=Inter:wght@400;500;600;700&family=JetBrains+Mono:wght@500;700&family=Noto+Serif+SC:wght@600;700&family=Noto+Sans+SC:wght@400;500;700&display=swap" rel="stylesheet">
|
||||
<script src="https://cdn.jsdelivr.net/npm/mermaid@10/dist/mermaid.min.js"></script>
|
||||
<style>
|
||||
:root{
|
||||
--bg:#F7F9FC; --card:#FFFFFF; --ink:#1E2A44; --ink2:#475569; --mute:#94A3B8;
|
||||
--line:#E6EBF2; --indigo:#4F6BED; --coral:#F26B5E; --mint:#2EC4B6;
|
||||
--t-blue:#E8EEFF; --t-coral:#FFEDEA; --t-mint:#E6F7F4; --t-amber:#FEF3C7;
|
||||
--drawer:#111A2E;
|
||||
--serif:"Cormorant Garamond","Noto Serif SC",Georgia,serif;
|
||||
--sans:"Inter","Noto Sans SC",-apple-system,"PingFang SC","Microsoft YaHei",sans-serif;
|
||||
--mono:"JetBrains Mono",ui-monospace,SFMono-Regular,Menlo,Consolas,monospace;
|
||||
--shadow:0 1px 2px rgba(30,42,68,.04),0 6px 18px rgba(30,42,68,.06);
|
||||
--shadow-sm:0 1px 2px rgba(30,42,68,.06);
|
||||
}
|
||||
*{box-sizing:border-box;margin:0;padding:0}
|
||||
html,body{height:100%;margin:0;overflow:hidden}
|
||||
body{font-family:var(--sans);color:var(--ink);background:var(--bg);
|
||||
background-image:radial-gradient(1200px 480px at 80% -8%,#EAF0FF 0%,rgba(234,240,255,0) 60%),
|
||||
radial-gradient(900px 420px at 0% 0%,#FFF1EE 0%,rgba(255,241,238,0) 55%);
|
||||
-webkit-font-smoothing:antialiased}
|
||||
.app{display:flex;flex-direction:column;height:100vh;overflow:hidden}
|
||||
|
||||
/* ========== HEADER ========== */
|
||||
header.bar{
|
||||
display:flex;align-items:center;gap:18px;padding:0 22px;
|
||||
min-height:62px;height:auto;
|
||||
background:rgba(255,255,255,.86);backdrop-filter:blur(8px);
|
||||
border-bottom:1px solid var(--line);position:relative;z-index:5;
|
||||
overflow:visible;flex-shrink:0;
|
||||
}
|
||||
.brand{display:flex;align-items:center;gap:10px;min-width:0;flex-shrink:0}
|
||||
.brand .mark{width:30px;height:30px;border-radius:9px;background:linear-gradient(135deg,var(--indigo),#7C8CF5 60%,var(--coral));
|
||||
display:grid;place-items:center;color:#fff;font-family:var(--serif);font-weight:700;font-size:18px;box-shadow:var(--shadow-sm)}
|
||||
.brand .t1{font-family:var(--serif);font-weight:700;font-size:20px;line-height:1.1;letter-spacing:.2px}
|
||||
.brand .t2{font-size:10px;color:var(--mute);letter-spacing:.12em;text-transform:uppercase;line-height:1.2}
|
||||
.search{display:flex;align-items:center;gap:8px;margin-left:8px;flex:1;max-width:640px}
|
||||
.field{display:flex;align-items:center;gap:7px;background:var(--card);border:1px solid var(--line);
|
||||
border-radius:10px;padding:7px 11px;box-shadow:var(--shadow-sm);flex:1}
|
||||
.field .k{font-family:var(--mono);font-size:11px;color:var(--mute)}
|
||||
.field input{border:0;outline:0;font-family:var(--mono);font-size:13px;color:var(--ink);width:100%;background:transparent}
|
||||
.chip{display:inline-flex;align-items:center;gap:6px;font-size:12px;font-weight:600;padding:6px 11px;border-radius:999px;white-space:nowrap}
|
||||
.chip.agent{background:var(--t-mint);color:#0E7E72;border:1px solid #B6EBE3}
|
||||
.chip.agent::before{content:"";width:7px;height:7px;border-radius:50%;background:var(--mint);box-shadow:0 0 0 3px rgba(46,196,182,.18);animation:pulse 1.8s infinite}
|
||||
.btn-login{font-size:12px;font-weight:600;color:#fff;background:var(--indigo);border:0;border-radius:10px;padding:8px 15px;cursor:pointer;box-shadow:var(--shadow-sm)}
|
||||
.btn-hot{font-size:12px;font-weight:600;color:#fff;background:linear-gradient(135deg,var(--coral),#FF8A7A);text-decoration:none;border:0;border-radius:10px;padding:8px 14px;cursor:pointer;box-shadow:var(--shadow-sm);display:inline-flex;align-items:center;gap:5px;white-space:nowrap}
|
||||
.btn-hot:hover{filter:brightness(1.05)}
|
||||
@keyframes pulse{50%{opacity:.4}}
|
||||
|
||||
/* ========== MAIN GRID ========== */
|
||||
main.grid{
|
||||
display:grid;grid-template-columns:262px 1fr 332px;
|
||||
gap:16px;padding:16px;flex:1;min-height:0;overflow:hidden;
|
||||
}
|
||||
aside.left,aside.right{min-height:0;overflow:auto;display:flex;flex-direction:column;gap:12px;padding-right:2px}
|
||||
section.center{min-height:0;display:flex;flex-direction:column;gap:14px;overflow:hidden}
|
||||
|
||||
.panel{background:var(--card);border:1px solid var(--line);border-radius:14px;box-shadow:var(--shadow)}
|
||||
.panel .ph{padding:14px 16px 6px}
|
||||
.panel .ph h3{margin:0;font-size:13px;font-weight:700;letter-spacing:.02em}
|
||||
.panel .ph .sub{font-size:11px;color:var(--mute);margin-top:2px}
|
||||
.eyebrow{font-family:var(--mono);font-size:10px;letter-spacing:.16em;text-transform:uppercase;color:var(--mute)}
|
||||
|
||||
/* ========== SCENARIO CARDS ========== */
|
||||
.scn{display:flex;gap:11px;align-items:flex-start;padding:11px 12px;border:1px solid var(--line);border-radius:12px;
|
||||
background:var(--card);cursor:pointer;transition:.15s;position:relative;box-shadow:var(--shadow-sm)}
|
||||
.scn:hover{border-color:#C7D2FE;transform:translateY(-1px)}
|
||||
.scn.active{border-color:var(--indigo);background:linear-gradient(180deg,#F5F8FF,#fff)}
|
||||
.scn.active::after{content:"";position:absolute;top:10px;right:10px;width:7px;height:7px;border-radius:50%;background:var(--coral)}
|
||||
.scn.running{pointer-events:none;opacity:.7}
|
||||
.scn .ico{width:30px;height:30px;border-radius:9px;display:grid;place-items:center;flex:0 0 auto;font-size:15px}
|
||||
.scn .t{min-width:0}
|
||||
.scn .t b{font-size:13px;font-weight:600;display:block}
|
||||
.scn .t small{font-size:11px;color:var(--mute);line-height:1.35;display:block;margin-top:1px}
|
||||
.tint-blue .ico{background:var(--t-blue);color:var(--indigo)}
|
||||
.tint-coral .ico{background:var(--t-coral);color:var(--coral)}
|
||||
.tint-mint .ico{background:var(--t-mint);color:#0E7E72}
|
||||
.tint-amber .ico{background:var(--t-amber);color:#B45309}
|
||||
|
||||
/* ========== CENTER HERO / GRAPH ========== */
|
||||
.hero{flex:1;min-height:0;display:flex;flex-direction:column;overflow:hidden}
|
||||
.hero .ph{display:flex;align-items:center;justify-content:space-between;flex-shrink:0}
|
||||
.hero .ph .live{font-family:var(--mono);font-size:11px;color:var(--indigo);background:var(--t-blue);padding:4px 9px;border-radius:999px}
|
||||
.graph-wrap{flex:1;min-height:0;position:relative;margin:0 4px 4px;border-radius:10px;
|
||||
background:radial-gradient(420px 320px at 50% 42%,#F2F6FF 0%,rgba(247,249,252,0) 70%);overflow:hidden}
|
||||
svg.graph{width:100%;height:100%;display:block}
|
||||
.g-edge{stroke:#9DB4F0;stroke-opacity:.55;stroke-width:1}
|
||||
.g-node circle{stroke:#fff;stroke-width:2;transition:.2s}
|
||||
.g-node:hover circle{filter:brightness(1.08)}
|
||||
.g-node text{font-family:var(--sans);font-size:10px;font-weight:600;fill:var(--ink);pointer-events:none;text-anchor:middle}
|
||||
.glow{filter:drop-shadow(0 0 6px rgba(242,107,94,.55))}
|
||||
/* hero-overlay: ONLY visible during loading/empty, hidden after graph renders */
|
||||
.hero-overlay{position:absolute;inset:0;display:flex;align-items:center;justify-content:center;flex-direction:column;gap:8px;pointer-events:none;
|
||||
transition:opacity .3s}
|
||||
.hero-overlay.hidden{opacity:0;pointer-events:none}
|
||||
.hero-overlay .hint{font-family:var(--mono);font-size:12px;color:var(--mute);background:rgba(255,255,255,.8);padding:6px 12px;border-radius:999px}
|
||||
.spin{width:26px;height:26px;border-radius:50%;border:3px solid var(--t-blue);border-top-color:var(--indigo);animation:rot .8s linear infinite}
|
||||
@keyframes rot{to{transform:rotate(360deg)}}
|
||||
.center-extra{flex:0 0 auto;max-height:46%;display:flex;flex-direction:column;gap:10px;overflow:auto}
|
||||
.center-extra:empty{display:none}
|
||||
iframe.plot{width:100%;height:420px;border:0;border-radius:10px;background:#fff}
|
||||
.mmd-box{background:#fff;padding:8px;border-radius:10px;overflow:auto;max-height:420px}
|
||||
|
||||
/* Graph legend bar */
|
||||
.graph-legend{display:flex;align-items:center;gap:14px;padding:6px 16px 10px;flex-wrap:wrap;flex-shrink:0}
|
||||
.graph-legend .leg-item{display:flex;align-items:center;gap:5px;font-size:11px;color:var(--ink2)}
|
||||
.graph-legend .leg-dot{width:10px;height:10px;border-radius:50%;flex-shrink:0}
|
||||
.graph-legend .leg-note{font-size:11px;color:var(--mute);margin-left:auto}
|
||||
|
||||
/* ========== RIGHT COLUMN ========== */
|
||||
.metrics{display:grid;grid-template-columns:1fr 1fr;gap:10px;padding:6px 14px 14px}
|
||||
.tile{border:1px solid var(--line);border-radius:12px;padding:12px;background:#FCFDFE;box-shadow:var(--shadow-sm)}
|
||||
.tile .lab{display:flex;align-items:center;gap:6px;font-size:11px;color:var(--ink2);font-weight:500}
|
||||
.tile .lab .dot{width:8px;height:8px;border-radius:50%}
|
||||
.tile .val{font-family:var(--mono);font-weight:700;font-size:24px;line-height:1.1;margin-top:6px;color:var(--ink)}
|
||||
.tile .of{font-size:11px;color:var(--mute);margin-top:3px}
|
||||
.d-indigo{background:var(--indigo)} .d-mint{background:var(--mint)} .d-coral{background:var(--coral)} .d-amber{background:#F59E0B}
|
||||
|
||||
.findings{padding:4px 14px 14px;display:flex;flex-direction:column;gap:9px;max-height:330px;overflow:auto}
|
||||
.find{border-left:3px solid var(--indigo);padding:8px 11px;background:#F8FAFF;border-radius:0 9px 9px 0}
|
||||
.find.critical{border-color:var(--coral);background:#FFF5F3}
|
||||
.find.warning{border-color:#F59E0B;background:#FFFBEB}
|
||||
.find.success{border-color:var(--mint);background:#F0FBF9}
|
||||
.find .ft{font-size:11px;font-weight:700;color:var(--ink2);display:flex;justify-content:space-between;gap:8px}
|
||||
.find .fb{font-size:12px;color:var(--ink);margin-top:3px;line-height:1.45}
|
||||
.find .fs{font-size:11px;color:var(--mute);margin-top:3px}
|
||||
.bar{height:6px;border-radius:4px;background:var(--line);overflow:hidden;margin-top:6px}
|
||||
.bar > i{display:block;height:100%;background:linear-gradient(90deg,var(--indigo),#7C8CF5)}
|
||||
.empty{font-size:12px;color:var(--mute);padding:18px;text-align:center}
|
||||
.loading-brief{font-size:12px;color:var(--mute);padding:18px;text-align:center;display:flex;flex-direction:column;align-items:center;gap:8px}
|
||||
.loading-brief .spin{width:20px;height:20px;border-width:2px}
|
||||
|
||||
/* CTA button to view full result */
|
||||
.view-result-btn{display:block;margin:8px 14px 14px;padding:10px 18px;border:0;border-radius:12px;
|
||||
background:linear-gradient(135deg,var(--indigo),#7C8CF5);color:#fff;font-family:var(--sans);
|
||||
font-size:13px;font-weight:600;cursor:pointer;text-align:center;box-shadow:var(--shadow);
|
||||
transition:.15s}
|
||||
.view-result-btn:hover{transform:translateY(-1px);box-shadow:0 4px 14px rgba(79,107,237,.25)}
|
||||
|
||||
/* ========== COMMAND DRAWER ========== */
|
||||
footer.drawer{display:flex;align-items:center;gap:14px;padding:0 18px;background:var(--drawer);color:#CBD5E1;font-family:var(--mono);font-size:12px;
|
||||
flex-shrink:0;min-height:52px;height:52px}
|
||||
.dots{display:flex;gap:6px}
|
||||
.dots i{width:11px;height:11px;border-radius:50%;display:block}
|
||||
.drawer .prompt{color:var(--mint)}
|
||||
.drawer .cmd{flex:1;white-space:nowrap;overflow:hidden;text-overflow:ellipsis;color:#E2E8F0}
|
||||
.drawer .st{color:var(--mute);font-size:11px;white-space:nowrap}
|
||||
.drawer .st.running{color:var(--indigo);animation:pulse 1s infinite}
|
||||
|
||||
/* ========== SCROLLBARS ========== */
|
||||
aside::-webkit-scrollbar,.findings::-webkit-scrollbar,.center-extra::-webkit-scrollbar{width:8px;height:8px}
|
||||
::-webkit-scrollbar-thumb{background:#D7DEEA;border-radius:8px}
|
||||
::-webkit-scrollbar-track{background:transparent}
|
||||
|
||||
/* ========== RESPONSIVE ========== */
|
||||
@media (max-width:1180px){
|
||||
main.grid{grid-template-columns:230px 1fr;grid-template-rows:auto auto}
|
||||
aside.right{grid-column:1 / span 2;flex-direction:row;flex-wrap:wrap}
|
||||
aside.right .panel{flex:1;min-width:280px}
|
||||
}
|
||||
@media (max-width:760px){
|
||||
main.grid{grid-template-columns:1fr;overflow:auto}
|
||||
aside.left,aside.right{grid-column:auto}
|
||||
.search{display:none}
|
||||
}
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<div class="app">
|
||||
|
||||
<!-- ============ TOP BAR ============ -->
|
||||
<header class="bar">
|
||||
<div class="brand">
|
||||
<div class="mark">G</div>
|
||||
<div>
|
||||
<div class="t1">GitLink Research Atlas</div>
|
||||
<div class="t2">科研代码图谱智能体</div>
|
||||
</div>
|
||||
</div>
|
||||
<div class="search">
|
||||
<label class="field"><span class="k">repo</span>
|
||||
<input id="owner" value="mindspore-Ecosystem" placeholder="owner">
|
||||
<span class="k" style="color:#CBD5E1">/</span>
|
||||
<input id="repo" value="mindspore" placeholder="repo" style="max-width:150px">
|
||||
</label>
|
||||
<label class="field"><span class="k">keyword</span>
|
||||
<input id="keyword" value="deep learning,机器学习" placeholder="S2 用,逗号分隔">
|
||||
</label>
|
||||
</div>
|
||||
<a class="btn-hot" href="hotspot.html">🔥 热点追踪</a>
|
||||
<span class="chip agent">AI Agent online</span>
|
||||
<button class="btn-login" id="loginBtn">Login</button>
|
||||
</header>
|
||||
|
||||
<!-- ============ MAIN ============ -->
|
||||
<main class="grid">
|
||||
|
||||
<!-- LEFT: scenarios -->
|
||||
<aside class="left">
|
||||
<div class="panel">
|
||||
<div class="ph"><div class="eyebrow">Research Scenarios</div>
|
||||
<h3>科研场景</h3><div class="sub">点选任一场景,服务器终端即运行 gitlink-cli + Python 算法</div>
|
||||
</div>
|
||||
<div id="scenarios" style="padding:8px 12px 14px;display:flex;flex-direction:column;gap:9px"></div>
|
||||
</div>
|
||||
</aside>
|
||||
|
||||
<!-- CENTER: graph / viz -->
|
||||
<section class="center">
|
||||
<div class="panel hero">
|
||||
<div class="ph">
|
||||
<div><div class="eyebrow">Knowledge Graph</div><h3 id="heroTitle">科研知识图谱</h3></div>
|
||||
<span class="live" id="heroLive" style="display:none">● live build</span>
|
||||
</div>
|
||||
<div class="graph-wrap">
|
||||
<svg class="graph" id="graph" preserveAspectRatio="xMidYMid meet"></svg>
|
||||
<div class="hero-overlay hidden" id="heroOverlay">
|
||||
<div class="hint">选择左侧科研场景以启动分析</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
<div class="graph-legend" id="graphLegend"></div>
|
||||
<div class="center-extra" id="centerExtra"></div>
|
||||
</section>
|
||||
|
||||
<!-- RIGHT: metrics + report -->
|
||||
<aside class="right">
|
||||
<div class="panel">
|
||||
<div class="ph"><div class="eyebrow">Live Metrics</div><h3>科研指标</h3></div>
|
||||
<div class="metrics" id="metrics"></div>
|
||||
</div>
|
||||
<div class="panel" style="flex:1;min-height:0;display:flex;flex-direction:column">
|
||||
<div class="ph"><div class="eyebrow">AI Research Brief</div><h3 id="briefTitle">研究简报</h3></div>
|
||||
<div class="findings" id="findings"><div class="empty">运行场景后,这里展示创新点 / 风险预警 / 协作推荐等结构化结论</div></div>
|
||||
<button class="view-result-btn" id="viewResultBtn" style="display:none" target="_blank">查看完整结果 →</button>
|
||||
</div>
|
||||
</aside>
|
||||
|
||||
</main>
|
||||
|
||||
<!-- ============ DRAWER ============ -->
|
||||
<footer class="drawer">
|
||||
<div class="dots"><i style="background:#FF5F57"></i><i style="background:#FEBC2E"></i><i style="background:#28C840"></i></div>
|
||||
<span class="prompt">atlas ❯</span>
|
||||
<span class="cmd" id="cmd">gitlink-cli research --repo mindspore-Ecosystem/mindspore --agent atlas</span>
|
||||
<span class="st" id="cmdStatus">ready</span>
|
||||
</footer>
|
||||
|
||||
</div>
|
||||
|
||||
<script>
|
||||
mermaid.initialize({startOnLoad:false, theme:"base", securityLevel:"loose",
|
||||
themeVariables:{primaryColor:"#E8EEFF",primaryTextColor:"#1E2A44",primaryBorderColor:"#4F6BED",lineColor:"#9DB4F0",fontSize:"12px"}});
|
||||
|
||||
let SCENARIOS=[], ACTIVE=null, LAST_RESULT_KEY=null;
|
||||
|
||||
/* ---------- node color palette ---------- */
|
||||
const NODE_COLOR = {
|
||||
repo:"#4F6BED", scholar:"#3B82F6", team:"#6366F1", topic:"#06B6D4", method:"#6366F1",
|
||||
paper:"#8B5CF6", dataset:"#2EC4B6", model:"#4F6BED", experiment:"#F26B5E",
|
||||
reproduce:"#2EC4B6", license:"#64748B", trend:"#F26B5E", issue:"#F59E0B", pr:"#10B981",
|
||||
default:"#94A3B8"
|
||||
};
|
||||
const NODE_LABEL = {
|
||||
repo:"仓库", scholar:"学者", topic:"主题", paper:"论文", dataset:"数据集",
|
||||
model:"模型", experiment:"实验", reproduce:"复现", license:"许可证",
|
||||
trend:"趋势", issue:"Issue", pr:"PR"
|
||||
};
|
||||
|
||||
/* Default decorative graph — clean, ~12 nodes showing the research world */
|
||||
const DEMO = {
|
||||
nodes:[
|
||||
{id:"repo",type:"repo",label:"GitLink Repo",imp:true},
|
||||
{id:"scholar",type:"scholar",label:"Scholar"},
|
||||
{id:"topic",type:"topic",label:"Topic"},
|
||||
{id:"paper",type:"paper",label:"Paper"},
|
||||
{id:"dataset",type:"dataset",label:"Dataset"},
|
||||
{id:"model",type:"model",label:"Model"},
|
||||
{id:"exp",type:"experiment",label:"Experiment",imp:true},
|
||||
{id:"repr",type:"reproduce",label:"Reproduce"},
|
||||
{id:"lic",type:"license",label:"License"},
|
||||
{id:"trend",type:"trend",label:"Trend"},
|
||||
{id:"issue",type:"issue",label:"Issue Signals"},
|
||||
{id:"pr",type:"pr",label:"Pull Request"}
|
||||
],
|
||||
edges:[["repo","scholar"],["repo","topic"],["repo","paper"],["repo","dataset"],
|
||||
["repo","model"],["repo","exp"],["repo","repr"],["repo","lic"],
|
||||
["scholar","paper"],["topic","trend"],["exp","repr"],["issue","pr"],["pr","repo"],["model","dataset"]]
|
||||
};
|
||||
|
||||
/* ---------- build legend HTML ---------- */
|
||||
function buildLegend(){
|
||||
const el=document.getElementById("graphLegend");
|
||||
const types=new Set();
|
||||
// gather types from current graph data
|
||||
const src=ACTIVE&&ACTIVE._graphData?ACTIVE._graphData:DEMO;
|
||||
(src.nodes||[]).forEach(n=>types.add(n.type||"default"));
|
||||
let h="";
|
||||
for(const t of [...types].sort()){
|
||||
h+=`<span class="leg-item"><span class="leg-dot" style="background:${NODE_COLOR[t]||NODE_COLOR.default}"></span>${NODE_LABEL[t]||t}</span>`;
|
||||
}
|
||||
h+=`<span class="leg-note">示例:科研仓库与学者/主题/数据集的关联 — 选择 S2 生成真实领域图谱</span>`;
|
||||
el.innerHTML=h;
|
||||
}
|
||||
|
||||
/* ---------- force-directed layout ---------- */
|
||||
function forceLayout(nodes, edges, W, H, iters){
|
||||
const cx=W/2, cy=H/2;
|
||||
nodes.forEach(n=>{ if(n.x==null){n.x=cx+(Math.random()-.5)*W*.5; n.y=cy+(Math.random()-.5)*H*.5;} });
|
||||
for(let it=0; it<iters; it++){
|
||||
for(const n of nodes){ n._fx=0; n._fy=0; }
|
||||
for(let i=0;i<nodes.length;i++){
|
||||
for(let j=i+1;j<nodes.length;j++){
|
||||
let dx=nodes[i].x-nodes[j].x, dy=nodes[i].y-nodes[j].y;
|
||||
let d2=dx*dx+dy*dy+0.02, d=Math.sqrt(d2), f=2600/d2;
|
||||
let fx=dx/d*f, fy=dy/d*f;
|
||||
nodes[i]._fx+=fx; nodes[i]._fy+=fy; nodes[j]._fx-=fx; nodes[j]._fy-=fy;
|
||||
}
|
||||
}
|
||||
for(const e of edges){
|
||||
const a=nodes.find(n=>n.id===e[0]), b=nodes.find(n=>n.id===e[1]); if(!a||!b) continue;
|
||||
let dx=b.x-a.x, dy=b.y-a.y, d=Math.sqrt(dx*dx+dy*dy)+.02, L=a.imp||b.imp?120:78, k=.04;
|
||||
let f=(d-L)*k, fx=dx/d*f, fy=dy/d*f;
|
||||
a._fx+=fx; a._fy+=fy; b._fx-=fx; b._fy-=fy;
|
||||
}
|
||||
for(const n of nodes){
|
||||
n._fx+=(cx-n.x)*.012; n._fy+=(cy-n.y)*.012;
|
||||
n.x+=Math.max(-14,Math.min(14,n._fx)); n.y+=Math.max(-14,Math.min(14,n._fy));
|
||||
n.x=Math.max(40,Math.min(W-40,n.x)); n.y=Math.max(34,Math.min(H-30,n.y));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
function drawGraph(data){
|
||||
const svg=document.getElementById("graph");
|
||||
const W=svg.clientWidth||760, H=svg.clientHeight||460;
|
||||
svg.setAttribute("viewBox",`0 0 ${W} ${H}`);
|
||||
let nodes=(data.nodes||[]).map(n=>Object.assign({},n));
|
||||
let rawEdges=(data.edges||[]);
|
||||
// normalize edges to [source,target] pairs
|
||||
let edges=rawEdges.map(e=>Array.isArray(e)?e:[e.source||e.from,e.target||e.to]);
|
||||
// cap to ~40 for performance
|
||||
if(nodes.length>40){
|
||||
const deg=new Map(nodes.map(n=>[n.id,0]));
|
||||
edges.forEach(e=>{deg.set(e[0],(deg.get(e[0])||0)+1);deg.set(e[1],(deg.get(e[1])||0)+1);});
|
||||
nodes.sort((a,b)=>(deg.get(b.id)||0)-(deg.get(a.id)||0));
|
||||
const keep=new Set(nodes.slice(0,40).map(n=>n.id));
|
||||
nodes=nodes.filter(n=>keep.has(n.id));
|
||||
edges=edges.filter(e=>keep.has(e[0])&&keep.has(e[1]));
|
||||
}
|
||||
forceLayout(nodes, edges, W, H, nodes.length>20?220:320);
|
||||
let s=`<defs>
|
||||
<filter id="glow"><feGaussianBlur stdDeviation="4" result="blur"/><feMerge><feMergeNode in="blur"/><feMergeNode in="SourceGraphic"/></feMerge></filter>
|
||||
</defs>`;
|
||||
for(const e of edges){
|
||||
const a=nodes.find(n=>n.id===e[0]), b=nodes.find(n=>n.id===e[1]); if(!a||!b) continue;
|
||||
s+=`<line class="g-edge" x1="${a.x.toFixed(1)}" y1="${a.y.toFixed(1)}" x2="${b.x.toFixed(1)}" y2="${b.y.toFixed(1)}"/>`;
|
||||
}
|
||||
for(const n of nodes){
|
||||
const col=NODE_COLOR[n.type]||NODE_COLOR.default;
|
||||
const r=n.imp?11:7;
|
||||
const glowAttr=n.imp?` filter="url(#glow)"`:"";
|
||||
s+=`<g class="g-node" transform="translate(${n.x.toFixed(1)},${n.y.toFixed(1)})">
|
||||
<circle r="${r}" fill="${col}"${glowAttr}/><text y="${r+12}">${esc(n.label||n.id).slice(0,16)}</text></g>`;
|
||||
}
|
||||
svg.innerHTML=s;
|
||||
// BUG FIX #1: after graph renders, immediately hide the overlay
|
||||
hideOverlay();
|
||||
buildLegend();
|
||||
}
|
||||
|
||||
/* ---------- overlay helpers ---------- */
|
||||
const overlay=document.getElementById("heroOverlay");
|
||||
function showOverlay(html){
|
||||
overlay.innerHTML=html||'<div class="hint">选择左侧科研场景以启动分析</div>';
|
||||
overlay.classList.remove("hidden");
|
||||
}
|
||||
function showLoading(msg){
|
||||
overlay.innerHTML=`<div class="spin"></div><div class="hint">${esc(msg||"加载中...")}</div>`;
|
||||
overlay.classList.remove("hidden");
|
||||
}
|
||||
function hideOverlay(){
|
||||
overlay.classList.add("hidden");
|
||||
}
|
||||
|
||||
/* ---------- init scenarios ---------- */
|
||||
async function init(){
|
||||
try{
|
||||
const r=await fetch("/api/scenarios",{headers:{"X-Demo-Token":token()}});
|
||||
const j=await r.json(); SCENARIOS=j.scenarios||[];
|
||||
}catch(e){ SCENARIOS=[]; }
|
||||
const tints=["tint-blue","tint-coral","tint-mint","tint-amber","tint-blue","tint-mint"];
|
||||
const icons=["Ⅰ","Ⅱ","Ⅲ","Ⅳ","Ⅴ","Ⅵ"];
|
||||
const box=document.getElementById("scenarios");
|
||||
box.innerHTML=SCENARIOS.map((s,i)=>`
|
||||
<div class="scn ${tints[i%tints.length]}" data-key="${s.key}">
|
||||
<div class="ico">${icons[i]||"•"}</div>
|
||||
<div class="t"><b>${esc(s.label)}</b><small>${esc(s.desc)}<br><span style="color:#A5B4CF">需: ${(s.needs||[]).join(" · ")}</span></small></div>
|
||||
</div>`).join("");
|
||||
box.querySelectorAll(".scn").forEach(el=>el.onclick=()=>run(el.dataset.key, el));
|
||||
|
||||
// Draw default graph
|
||||
drawGraph(DEMO);
|
||||
setMetrics(defaultMetrics());
|
||||
buildLegend();
|
||||
|
||||
// Login button
|
||||
document.getElementById("loginBtn").onclick=()=>{
|
||||
const t=prompt("粘贴演示 token(留空表示无鉴权):", localStorage.getItem("demo_token")||"");
|
||||
if(t!=null){ localStorage.setItem("demo_token",t.trim()); alert(t.trim()?"已保存":"已清除"); }
|
||||
};
|
||||
|
||||
// View result button
|
||||
document.getElementById("viewResultBtn").onclick=()=>{
|
||||
if(LAST_RESULT_KEY) window.open("result.html?scenario="+LAST_RESULT_KEY,"_blank");
|
||||
};
|
||||
}
|
||||
|
||||
/* ---------- run scenario ---------- */
|
||||
async function run(key, el){
|
||||
const s=SCENARIOS.find(x=>x.key===key); if(!s) return;
|
||||
ACTIVE=s;
|
||||
// Mark card active + running
|
||||
document.querySelectorAll(".scn").forEach(e=>{e.classList.remove("active");e.classList.remove("running");});
|
||||
if(el) el.classList.add("active","running");
|
||||
document.getElementById("heroTitle").textContent=s.label;
|
||||
// BUG FIX #3: immediate loading feedback everywhere
|
||||
showLoading("服务器终端执行中:"+s.script+" …");
|
||||
document.getElementById("heroLive").style.display="none";
|
||||
document.getElementById("cmd").textContent=cmdText(s);
|
||||
document.getElementById("cmdStatus").textContent="running…";
|
||||
document.getElementById("cmdStatus").className="st running";
|
||||
document.getElementById("centerExtra").innerHTML="";
|
||||
document.getElementById("viewResultBtn").style.display="none";
|
||||
LAST_RESULT_KEY=null;
|
||||
|
||||
// BUG FIX #3: set brief + metrics to loading state
|
||||
setFindings('<div class="loading-brief"><div class="spin"></div><div>正在分析…</div></div>');
|
||||
setMetrics(tile("indigo","status","…","计算中")+tile("mint","status","…","")+tile("coral","status","…","")+tile("amber","status","…",""));
|
||||
|
||||
const body={scenario:key, owner:val("owner"), repo:val("repo"), keyword:val("keyword")};
|
||||
try{
|
||||
const r=await fetch("/api/run",{method:"POST",headers:{"Content-Type":"application/json","X-Demo-Token":token()},body:JSON.stringify(body)});
|
||||
const j=await r.json();
|
||||
// Clear running state
|
||||
document.querySelectorAll(".scn").forEach(e=>e.classList.remove("running"));
|
||||
document.getElementById("cmdStatus").textContent=(j.ok?"✓ done":"✗ failed")+" · "+(j.duration||"");
|
||||
document.getElementById("cmdStatus").className="st";
|
||||
if(!j.ok){
|
||||
document.getElementById("heroOverlay").innerHTML='<div class="hint" style="color:#F26B5E">运行失败:'+esc(j.error||"未知错误")+'</div>';
|
||||
document.getElementById("heroOverlay").classList.remove("hidden");
|
||||
return;
|
||||
}
|
||||
renderResult(s, j);
|
||||
}catch(e){
|
||||
document.querySelectorAll(".scn").forEach(e=>e.classList.remove("running"));
|
||||
document.getElementById("cmdStatus").textContent="✗ 请求失败";
|
||||
document.getElementById("cmdStatus").className="st";
|
||||
document.getElementById("heroOverlay").innerHTML='<div class="hint" style="color:#F26B5E">请求失败:'+esc(e)+'</div>';
|
||||
document.getElementById("heroOverlay").classList.remove("hidden");
|
||||
}
|
||||
}
|
||||
|
||||
function cmdText(s){
|
||||
const o=val("owner"), r=val("repo"), k=val("keyword");
|
||||
if(s.key==="s2") return `gitlink-cli research --graph -k "${k}"`;
|
||||
return `gitlink-cli research ${s.script.replace(".py","")} --repo ${o}/${r}`;
|
||||
}
|
||||
function val(id){return document.getElementById(id).value.trim();}
|
||||
function token(){return localStorage.getItem("demo_token")||"";}
|
||||
|
||||
/* ---------- render result ---------- */
|
||||
function parseJson(j){
|
||||
if(!j.artifacts||!j.artifacts.json) return null;
|
||||
try{return JSON.parse(j.artifacts.json);}catch(e){return null;}
|
||||
}
|
||||
|
||||
function renderResult(s, j){
|
||||
const obj=parseJson(j);
|
||||
const extra=document.getElementById("centerExtra");
|
||||
|
||||
// S2: real knowledge graph
|
||||
if(obj && obj.nodes && obj.edges){
|
||||
const graphData={nodes:obj.nodes.map(n=>({id:n.id,type:n.type||"default",label:n.label||n.id,imp:false})), edges:obj.edges.map(e=>[e.source||e.from,e.target||e.to])};
|
||||
drawGraph(graphData);
|
||||
ACTIVE._graphData=graphData;
|
||||
document.getElementById("heroLive").style.display="";
|
||||
document.getElementById("heroLive").textContent="● live — "+(obj.nodes.length)+" nodes, "+(obj.edges.length)+" edges";
|
||||
}
|
||||
// S6: plotly iframe
|
||||
else if(j.artifacts && j.artifacts.html){
|
||||
const blob=URL.createObjectURL(new Blob([j.artifacts.html],{type:"text/html"}));
|
||||
extra.innerHTML='<div class="panel" style="padding:10px"><div class="eyebrow" style="padding:4px 6px">S6 Visualization · plotly</div>'+
|
||||
'<iframe class="plot" src="'+blob+'"></iframe></div>';
|
||||
// restore demo graph if no real graph
|
||||
drawGraph(DEMO);
|
||||
ACTIVE._graphData=null;
|
||||
}
|
||||
// S1/S4: mermaid
|
||||
else if(j.artifacts && j.artifacts.mermaid){
|
||||
const id="mmd_"+Date.now();
|
||||
extra.innerHTML='<div class="panel" style="padding:10px"><div class="eyebrow" style="padding:4px 6px">关系图 · Mermaid</div><div class="mmd-box"><div class="mermaid" id="'+id+'">'+j.artifacts.mermaid.replace(/```mermaid|```/g,"")+'</div></div></div>';
|
||||
setTimeout(()=>mermaid.run({nodes:["#"+id]}).catch(()=>{}),30);
|
||||
drawGraph(DEMO);
|
||||
ACTIVE._graphData=null;
|
||||
}
|
||||
// S3/S5 or others: keep demo graph, hide overlay
|
||||
else{
|
||||
drawGraph(DEMO);
|
||||
ACTIVE._graphData=null;
|
||||
}
|
||||
|
||||
// Metrics + findings
|
||||
if(obj){
|
||||
setMetrics(metricsFor(s.key, obj));
|
||||
setFindings(findingsFor(s.key, obj));
|
||||
} else {
|
||||
setMetrics(defaultMetrics());
|
||||
setFindings('<div class="empty">'+(j.stdout?esc(j.stdout.slice(-400)):"(无可解析 JSON 产物)")+'</div>');
|
||||
}
|
||||
|
||||
// Show "view full result" CTA
|
||||
LAST_RESULT_KEY=s.key;
|
||||
const btn=document.getElementById("viewResultBtn");
|
||||
btn.style.display="block";
|
||||
btn.textContent="查看完整结果 → "+s.label;
|
||||
}
|
||||
|
||||
/* ---------- metrics per scenario ---------- */
|
||||
function tile(c,label,val,of){return `<div class="tile"><div class="lab"><span class="dot d-${c}"></span>${label}</div><div class="val">${val}</div><div class="of">${of||""}</div></div>`;}
|
||||
function defaultMetrics(){
|
||||
return tile("indigo","commits","1,248","+18% this month")+
|
||||
tile("mint","PR merge","72.6%","healthy collab")+
|
||||
tile("indigo","contributors","36","8 core scholars")+
|
||||
tile("coral","compliance","88","示例数据 · 待运行");
|
||||
}
|
||||
function metricsFor(key,o){
|
||||
const M=o.meta||{};
|
||||
switch(key){
|
||||
case "s1": return tile("indigo","commits",M.commit_count||0,"采样提交")+tile("mint","merged PR",M.merged_pr_count||0,"已合并")+tile("amber","docs",M.doc_count||0,"文档文件")+tile("coral","创新点",(o.innovation_points||[]).length,"高影响合并");
|
||||
case "s2": return tile("indigo","repos",M.repo_count||(o.nodes||[]).length,"仓库节点")+tile("coral","scholars",M.scholar_count||0,"学者")+tile("mint","topics",M.topic_count||0,"主题方向")+tile("amber","edges",M.edge_count||(o.edges||[]).length,"关系边");
|
||||
case "s3": return tile("mint","repro",(o.repro_score||0)+"/10","复现性")+tile("indigo","compliance",(o.compliance_score||0)+"/10","合规")+tile("coral","risks",(o.risks||o.risk_items||[]).length,"风险项")+tile("amber","license",esc(o.license?(typeof o.license==="string"?o.license:(o.license.id||o.license.name||o.license.spdx_id||"—")):(o.license_id||"—")),"识别许可证");
|
||||
case "s4": return tile("indigo","gap topics",(o.gap_topics||[]).length,"缺口主题")+tile("mint","candidates",(o.candidates||[]).length,"推荐候选")+tile("coral","top score",(o.candidates&&o.candidates[0]?o.candidates[0].score:0),"最高匹配")+tile("amber","pool",o.meta&&o.meta.pool_size||0,"候选池");
|
||||
case "s5": {const tw=(o.week_stats&&o.week_stats.this_week)||{}; return tile("indigo","commits/周",tw.commits||0,"本周")+tile("coral","risks",(o.risk_warnings||[]).length,"预警")+tile("mint","PR 合并",tw.prs_merged||0,"本周")+tile("amber","trend",(o.trend&&o.trend.activity_level)||"—","活跃度");}
|
||||
case "s6": {const tl=o.timeline||{}; const sm=a=>(a||[]).reduce((x,y)=>x+(y||0),0); return tile("indigo","weeks",(tl.labels||[]).length,"时间线周数")+tile("coral","commits",sm(tl.commits),"累计")+tile("mint","PRs",sm(tl.prs),"累计")+tile("amber","issues",sm(tl.issues),"累计");}
|
||||
}
|
||||
return defaultMetrics();
|
||||
}
|
||||
function setMetrics(html){document.getElementById("metrics").innerHTML=html;}
|
||||
|
||||
/* ---------- findings per scenario ---------- */
|
||||
function find(cls,title,body,sub){return `<div class="find ${cls}"><div class="ft"><span>${title}</span></div><div class="fb">${body}</div>${sub?`<div class="fs">${sub}</div>`:""}</div>`;}
|
||||
function rank(c){return `<div class="bar"><i style="width:${Math.min(100,Math.max(4,c))}%"></i></div>`;}
|
||||
function findingsFor(key,o){
|
||||
let out="";
|
||||
switch(key){
|
||||
case "s1": (o.innovation_points||[]).slice(0,6).forEach(i=>{out+=find("warning","["+i.category+"] 创新点",esc(i.description),esc(i.evidence));}); break;
|
||||
case "s2": (o.topic_heat||[]).slice(0,4).forEach(h=>{out+=find("success","主题热度 · "+h.topic,"覆盖 "+h.count+" 个仓库",rank(h.count*15));});
|
||||
(o.core_scholars||[]).slice(0,3).forEach(s=>{out+=find("",esc(s.login),"关联 "+s.repo_count+" 个仓库");}); break;
|
||||
case "s3": (o.risk_items||o.risks||[]).slice(0,6).forEach(r=>{const lv=(r.level||"").includes("high")||r.severity==="P0"?"critical":"warning"; out+=find(lv,esc((r.level||r.severity||"")+" · "+(r.category||r.name||"")),esc(r.detail||r.evidence||r.name||""),esc(r.file?r.file:""));});
|
||||
out+=find("success","复现性 "+(o.repro_score||0)+"/10 · 合规 "+(o.compliance_score||0)+"/10","许可证: "+esc(o.license?(typeof o.license==="string"?o.license:(o.license.id||"—")):"—")); break;
|
||||
case "s4": (o.candidates||[]).slice(0,5).forEach((c,i)=>{out+=find(i===0?"success":"", "#"+(i+1)+" "+esc(c.login), (c.reasons||[]).join(";")||"—", rank(c.score)+` <span style="float:right">${c.score}分</span>`);}); break;
|
||||
case "s5": (o.risk_warnings||[]).slice(0,6).forEach(w=>{out+=find(w.level==="critical"?"critical":"warning",esc(w.level+" · "+w.type),esc(w.message),esc(w.suggestion||""));}); break;
|
||||
case "s6": out+=find("success","可视化产物","时间线 "+((o.timeline&&o.timeline.labels||[]).length)+" 周 · 语言 "+((o.language_pie&&o.language_pie.length)||0)+" 类","已生成可交互 plotly 图表"); break;
|
||||
}
|
||||
return out||'<div class="empty">(该场景无结构化简报,详见完整结果页)</div>';
|
||||
}
|
||||
function setFindings(html){document.getElementById("findings").innerHTML=html;}
|
||||
|
||||
/* ---------- utilities ---------- */
|
||||
function esc(s){return String(s).replace(/[&<>]/g,c=>({'&':'&','<':'<','>':'>'}[c]));}
|
||||
|
||||
/* ---------- resize handler ---------- */
|
||||
let resizeTimer;
|
||||
window.addEventListener("resize",()=>{
|
||||
clearTimeout(resizeTimer);
|
||||
resizeTimer=setTimeout(()=>{
|
||||
const src=ACTIVE&&ACTIVE._graphData?ACTIVE._graphData:DEMO;
|
||||
drawGraph(src);
|
||||
},200);
|
||||
});
|
||||
|
||||
/* go */
|
||||
init();
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
|
|
@ -0,0 +1,504 @@
|
|||
<!DOCTYPE html>
|
||||
<html lang="zh-CN">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
||||
<title>GitLink Research Atlas · 结果详情</title>
|
||||
<link rel="preconnect" href="https://fonts.googleapis.com">
|
||||
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
|
||||
<link href="https://fonts.googleapis.com/css2?family=Cormorant+Garamond:wght@500;600;700&family=Inter:wght@400;500;600;700&family=JetBrains+Mono:wght@500;700&family=Noto+Serif+SC:wght@600;700&family=Noto+Sans+SC:wght@400;500;700&display=swap" rel="stylesheet">
|
||||
<script src="https://cdn.jsdelivr.net/npm/mermaid@10/dist/mermaid.min.js"></script>
|
||||
<style>
|
||||
:root{
|
||||
--bg:#F7F9FC; --card:#FFFFFF; --ink:#1E2A44; --ink2:#475569; --mute:#94A3B8;
|
||||
--line:#E6EBF2; --indigo:#4F6BED; --coral:#F26B5E; --mint:#2EC4B6;
|
||||
--t-blue:#E8EEFF; --t-coral:#FFEDEA; --t-mint:#E6F7F4; --t-amber:#FEF3C7;
|
||||
--serif:"Cormorant Garamond","Noto Serif SC",Georgia,serif;
|
||||
--sans:"Inter","Noto Sans SC",-apple-system,"PingFang SC","Microsoft YaHei",sans-serif;
|
||||
--mono:"JetBrains Mono",ui-monospace,SFMono-Regular,Menlo,Consolas,monospace;
|
||||
--shadow:0 1px 2px rgba(30,42,68,.04),0 6px 18px rgba(30,42,68,.06);
|
||||
--shadow-sm:0 1px 2px rgba(30,42,68,.06);
|
||||
}
|
||||
*{box-sizing:border-box;margin:0;padding:0}
|
||||
html,body{height:auto;min-height:100%;margin:0}
|
||||
body{font-family:var(--sans);color:var(--ink);background:var(--bg);
|
||||
background-image:radial-gradient(1200px 480px at 80% -8%,#EAF0FF 0%,rgba(234,240,255,0) 60%),
|
||||
radial-gradient(900px 420px at 0% 0%,#FFF1EE 0%,rgba(255,241,238,0) 55%);
|
||||
-webkit-font-smoothing:antialiased}
|
||||
|
||||
/* ========== PRINT ========== */
|
||||
@media print{
|
||||
body{background:#fff!important}
|
||||
.no-print{display:none!important}
|
||||
.page-wrap{max-width:100%!important;box-shadow:none!important}
|
||||
}
|
||||
|
||||
/* ========== PAGE WRAPPER ========== */
|
||||
.page-wrap{max-width:1120px;margin:0 auto;padding:32px 40px 60px}
|
||||
@media(max-width:760px){.page-wrap{padding:20px 16px 40px}}
|
||||
|
||||
/* ========== HEADER ========== */
|
||||
.res-header{margin-bottom:32px}
|
||||
.res-header .back{display:inline-flex;align-items:center;gap:6px;font-size:13px;font-weight:600;
|
||||
color:var(--indigo);text-decoration:none;margin-bottom:16px;padding:6px 14px;border-radius:10px;
|
||||
background:var(--t-blue);transition:.15s}
|
||||
.res-header .back:hover{background:#D6DEFF}
|
||||
.res-header h1{font-family:var(--serif);font-weight:700;font-size:34px;line-height:1.2;letter-spacing:.3px;color:var(--ink)}
|
||||
.res-header .sub{font-size:15px;color:var(--ink2);margin-top:6px;line-height:1.5}
|
||||
.res-header .meta{display:flex;flex-wrap:wrap;gap:16px;margin-top:14px;align-items:center}
|
||||
.res-header .meta .chip{display:inline-flex;align-items:center;gap:6px;font-size:12px;font-weight:500;
|
||||
padding:5px 12px;border-radius:999px;background:var(--card);border:1px solid var(--line);color:var(--ink2);box-shadow:var(--shadow-sm)}
|
||||
.res-header .meta .chip b{font-family:var(--mono);font-weight:700;color:var(--ink)}
|
||||
.res-header .cmd-drawer{margin-top:14px;background:var(--ink);color:#CBD5E1;font-family:var(--mono);
|
||||
font-size:12px;padding:10px 16px;border-radius:10px;display:flex;align-items:center;gap:10px;overflow:hidden}
|
||||
.res-header .cmd-drawer .dots{display:flex;gap:6px;flex-shrink:0}
|
||||
.res-header .cmd-drawer .dots i{width:11px;height:11px;border-radius:50%;display:block}
|
||||
.res-header .cmd-drawer .cmd{flex:1;white-space:nowrap;overflow:hidden;text-overflow:ellipsis;color:#E2E8F0}
|
||||
|
||||
/* ========== SECTION TITLES ========== */
|
||||
.section{margin-top:36px}
|
||||
.section-title{font-family:var(--serif);font-weight:700;font-size:22px;color:var(--ink);margin-bottom:16px;
|
||||
padding-bottom:8px;border-bottom:2px solid var(--line)}
|
||||
.section-title .eyebrow{font-family:var(--mono);font-size:10px;letter-spacing:.16em;text-transform:uppercase;
|
||||
color:var(--mute);display:block;margin-bottom:4px}
|
||||
|
||||
/* ========== METRICS ROW ========== */
|
||||
.metrics-row{display:grid;grid-template-columns:repeat(4,1fr);gap:14px;margin-bottom:8px}
|
||||
@media(max-width:760px){.metrics-row{grid-template-columns:repeat(2,1fr)}}
|
||||
.metric-card{background:var(--card);border:1px solid var(--line);border-radius:12px;padding:18px;
|
||||
box-shadow:var(--shadow);text-align:center}
|
||||
.metric-card .label{font-size:12px;color:var(--ink2);font-weight:500;margin-bottom:8px}
|
||||
.metric-card .value{font-family:var(--mono);font-weight:700;font-size:32px;color:var(--ink);line-height:1}
|
||||
.metric-card .note{font-size:11px;color:var(--mute);margin-top:6px}
|
||||
|
||||
/* ========== KNOWLEDGE GRAPH ========== */
|
||||
.graph-container{background:var(--card);border:1px solid var(--line);border-radius:14px;box-shadow:var(--shadow);
|
||||
padding:20px;overflow:hidden}
|
||||
.graph-svg-wrap{width:100%;height:500px;position:relative;background:radial-gradient(500px 380px at 50% 45%,#F2F6FF 0%,rgba(255,255,255,0) 70%);
|
||||
border-radius:10px;overflow:hidden}
|
||||
.graph-svg-wrap svg{width:100%;height:100%;display:block}
|
||||
.g-edge{stroke:#9DB4F0;stroke-opacity:.55;stroke-width:1}
|
||||
.g-node circle{stroke:#fff;stroke-width:2;transition:.2s}
|
||||
.g-node:hover circle{filter:brightness(1.08)}
|
||||
.g-node text{font-family:var(--sans);font-size:11px;font-weight:600;fill:var(--ink);pointer-events:none;text-anchor:middle}
|
||||
.graph-legend{display:flex;flex-wrap:wrap;gap:14px;margin-top:14px;padding-top:12px;border-top:1px solid var(--line)}
|
||||
.graph-legend .leg-item{display:flex;align-items:center;gap:5px;font-size:12px;color:var(--ink2)}
|
||||
.graph-legend .leg-dot{width:10px;height:10px;border-radius:50%;flex-shrink:0}
|
||||
.graph-legend .leg-note{font-size:12px;color:var(--mute);margin-left:auto}
|
||||
|
||||
/* ========== VIZ IFRAME (S6) ========== */
|
||||
.viz-container{background:var(--card);border:1px solid var(--line);border-radius:14px;box-shadow:var(--shadow);
|
||||
padding:12px;overflow:hidden}
|
||||
.viz-container iframe{width:100%;height:520px;border:0;border-radius:10px;background:#fff}
|
||||
|
||||
/* ========== MERMAID ========== */
|
||||
.mmd-container{background:var(--card);border:1px solid var(--line);border-radius:14px;box-shadow:var(--shadow);
|
||||
padding:16px;overflow:auto;max-height:600px}
|
||||
|
||||
/* ========== RADAR / SCORE (S3) ========== */
|
||||
.score-display{display:flex;align-items:center;justify-content:center;gap:40px;padding:30px;
|
||||
background:var(--card);border:1px solid var(--line);border-radius:14px;box-shadow:var(--shadow);flex-wrap:wrap}
|
||||
.score-circle{width:140px;height:140px;border-radius:50%;display:flex;flex-direction:column;align-items:center;justify-content:center;
|
||||
position:relative;border:6px solid}
|
||||
.score-circle .val{font-family:var(--mono);font-weight:700;font-size:40px;line-height:1}
|
||||
.score-circle .label{font-size:13px;color:var(--ink2);margin-top:2px}
|
||||
.score-circle.repro{border-color:var(--mint);color:var(--mint);background:var(--t-mint)}
|
||||
.score-circle.compliance{border-color:var(--indigo);color:var(--indigo);background:var(--t-blue)}
|
||||
|
||||
/* ========== WEEKLY COMPARISON (S5) ========== */
|
||||
.weekly-bars{display:flex;flex-direction:column;gap:16px;padding:20px;
|
||||
background:var(--card);border:1px solid var(--line);border-radius:14px;box-shadow:var(--shadow)}
|
||||
.week-row{display:flex;align-items:center;gap:14px}
|
||||
.week-row .label{width:70px;font-size:13px;font-weight:600;color:var(--ink2);flex-shrink:0}
|
||||
.week-row .bar-bg{flex:1;height:28px;background:#F1F5F9;border-radius:8px;overflow:hidden;position:relative}
|
||||
.week-row .bar-fill{height:100%;border-radius:8px;transition:width .6s ease}
|
||||
.week-row .bar-fill.this-week{background:linear-gradient(90deg,var(--indigo),#7C8CF5)}
|
||||
.week-row .bar-fill.last-week{background:var(--line)}
|
||||
.week-row .bar-val{position:absolute;right:8px;top:50%;transform:translateY(-50%);font-family:var(--mono);font-size:12px;font-weight:600;color:var(--ink)}
|
||||
.risk-list{margin-top:20px;display:flex;flex-direction:column;gap:10px}
|
||||
.risk-item{padding:10px 14px;border-radius:10px;border-left:4px solid}
|
||||
.risk-item.critical{border-color:var(--coral);background:var(--t-coral)}
|
||||
.risk-item.warning{border-color:#F59E0B;background:var(--t-amber)}
|
||||
.risk-item .type{font-size:11px;font-family:var(--mono);color:var(--mute);text-transform:uppercase}
|
||||
.risk-item .msg{font-size:13px;color:var(--ink);margin-top:4px}
|
||||
.risk-item .suggestion{font-size:12px;color:var(--ink2);margin-top:4px;font-style:italic}
|
||||
|
||||
/* ========== REPORT (markdown) ========== */
|
||||
.report-content{background:var(--card);border:1px solid var(--line);border-radius:14px;box-shadow:var(--shadow);
|
||||
padding:28px 32px;line-height:1.75;font-size:14px;color:var(--ink)}
|
||||
.report-content h1{font-family:var(--serif);font-weight:700;font-size:26px;margin:24px 0 12px;padding-bottom:8px;border-bottom:2px solid var(--line)}
|
||||
.report-content h1:first-child{margin-top:0}
|
||||
.report-content h2{font-family:var(--serif);font-weight:700;font-size:20px;margin:20px 0 10px;color:var(--ink)}
|
||||
.report-content h3{font-weight:700;font-size:16px;margin:16px 0 8px;color:var(--ink2)}
|
||||
.report-content p{margin:10px 0}
|
||||
.report-content ul,.report-content ol{margin:8px 0;padding-left:24px}
|
||||
.report-content li{margin:4px 0}
|
||||
.report-content strong{font-weight:700;color:var(--ink)}
|
||||
.report-content code{font-family:var(--mono);font-size:12.5px;background:var(--t-blue);padding:2px 6px;border-radius:4px;color:var(--indigo)}
|
||||
.report-content table{width:100%;border-collapse:collapse;margin:12px 0;font-size:13px}
|
||||
.report-content th{background:var(--t-blue);font-weight:600;text-align:left;padding:8px 12px;border:1px solid var(--line)}
|
||||
.report-content td{padding:8px 12px;border:1px solid var(--line)}
|
||||
|
||||
/* ========== STRUCTURED CARDS ========== */
|
||||
.cards-grid{display:grid;grid-template-columns:repeat(auto-fill,minmax(320px,1fr));gap:16px;margin-top:20px}
|
||||
.conclusion-card{background:var(--card);border:1px solid var(--line);border-radius:12px;box-shadow:var(--shadow);
|
||||
padding:16px 18px;border-top:4px solid}
|
||||
.conclusion-card.innovation{border-top-color:var(--indigo)}
|
||||
.conclusion-card.risk{border-top-color:var(--coral)}
|
||||
.conclusion-card.collab{border-top-color:var(--mint)}
|
||||
.conclusion-card .card-title{font-weight:700;font-size:14px;color:var(--ink);margin-bottom:8px}
|
||||
.conclusion-card .card-body{font-size:13px;color:var(--ink2);line-height:1.55}
|
||||
.conclusion-card .card-body p{margin:4px 0}
|
||||
|
||||
/* ========== LOADING STATE ========== */
|
||||
.loading-state{display:flex;flex-direction:column;align-items:center;justify-content:center;padding:80px 20px;color:var(--mute);gap:16px}
|
||||
.spin-lg{width:40px;height:40px;border-radius:50%;border:4px solid var(--t-blue);border-top-color:var(--indigo);animation:rot .8s linear infinite}
|
||||
@keyframes rot{to{transform:rotate(360deg)}}
|
||||
|
||||
/* ========== ERROR ========== */
|
||||
.error-state{padding:60px 20px;text-align:center;color:var(--coral)}
|
||||
.error-state h2{font-family:var(--serif);font-size:24px;margin-bottom:12px}
|
||||
.error-state p{font-size:14px;color:var(--ink2);max-width:500px;margin:0 auto}
|
||||
|
||||
/* empty */
|
||||
.empty-state{padding:60px 20px;text-align:center;color:var(--mute)}
|
||||
.empty-state p{font-size:14px}
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<div class="page-wrap" id="app">
|
||||
<div class="loading-state"><div class="spin-lg"></div><div>正在加载结果…</div></div>
|
||||
</div>
|
||||
|
||||
<script>
|
||||
mermaid.initialize({startOnLoad:false, theme:"base", securityLevel:"loose",
|
||||
themeVariables:{primaryColor:"#E8EEFF",primaryTextColor:"#1E2A44",primaryBorderColor:"#4F6BED",lineColor:"#9DB4F0",fontSize:"13px"}});
|
||||
|
||||
const NODE_COLOR={repo:"#4F6BED",scholar:"#3B82F6",team:"#6366F1",topic:"#06B6D4",method:"#6366F1",
|
||||
paper:"#8B5CF6",dataset:"#2EC4B6",model:"#4F6BED",experiment:"#F26B5E",
|
||||
reproduce:"#2EC4B6",license:"#64748B",trend:"#F26B5E",issue:"#F59E0B",pr:"#10B981",default:"#94A3B8"};
|
||||
const NODE_LABEL={repo:"仓库",scholar:"学者",topic:"主题",paper:"论文",dataset:"数据集",
|
||||
model:"模型",experiment:"实验",reproduce:"复现",license:"许可证",trend:"趋势",issue:"Issue",pr:"PR"};
|
||||
|
||||
function esc(s){return String(s).replace(/[&<>]/g,c=>({'&':'&','<':'<','>':'>'}[c]));}
|
||||
function token(){return localStorage.getItem("demo_token")||"";}
|
||||
function licOf(o){const l=o.license;if(!l)return o.license_id||"—";return typeof l==="string"?l:(l.id||l.name||l.spdx_id||l.identifier||"—");}
|
||||
|
||||
/* ========== Force layout (same as index) ========== */
|
||||
function forceLayout(nodes,edges,W,H,iters){
|
||||
const cx=W/2,cy=H/2;
|
||||
nodes.forEach(n=>{if(n.x==null){n.x=cx+(Math.random()-.5)*W*.5;n.y=cy+(Math.random()-.5)*H*.5;}});
|
||||
for(let it=0;it<iters;it++){
|
||||
for(const n of nodes){n._fx=0;n._fy=0;}
|
||||
for(let i=0;i<nodes.length;i++){for(let j=i+1;j<nodes.length;j++){
|
||||
let dx=nodes[i].x-nodes[j].x,dy=nodes[i].y-nodes[j].y,d2=dx*dx+dy*dy+.02,d=Math.sqrt(d2),f=2600/d2,fx=dx/d*f,fy=dy/d*f;
|
||||
nodes[i]._fx+=fx;nodes[i]._fy+=fy;nodes[j]._fx-=fx;nodes[j]._fy-=fy;}}
|
||||
for(const e of edges){const a=nodes.find(n=>n.id===e[0]),b=nodes.find(n=>n.id===e[1]);if(!a||!b)continue;
|
||||
let dx=b.x-a.x,dy=b.y-a.y,d=Math.sqrt(dx*dx+dy*dy)+.02,L=a.imp||b.imp?130:85,k=.04;
|
||||
let f=(d-L)*k,fx=dx/d*f,fy=dy/d*f;a._fx+=fx;a._fy+=fy;b._fx-=fx;b._fy-=fy;}
|
||||
for(const n of nodes){n._fx+=(cx-n.x)*.012;n._fy+=(cy-n.y)*.012;
|
||||
n.x+=Math.max(-14,Math.min(14,n._fx));n.y+=Math.max(-14,Math.min(14,n._fy));
|
||||
n.x=Math.max(45,Math.min(W-45,n.x));n.y=Math.max(38,Math.min(H-34,n.y));}}
|
||||
}
|
||||
|
||||
function drawGraphIn(containerId,data){
|
||||
const svg=document.getElementById(containerId); if(!svg) return;
|
||||
const W=svg.clientWidth||800,H=svg.clientHeight||500;
|
||||
svg.setAttribute("viewBox",`0 0 ${W} ${H}`);
|
||||
let nodes=(data.nodes||[]).map(n=>Object.assign({},n));
|
||||
let edges=(data.edges||[]).map(e=>Array.isArray(e)?e:[e.source||e.from,e.target||e.to]);
|
||||
if(nodes.length>40){
|
||||
const deg=new Map(nodes.map(n=>[n.id,0]));
|
||||
edges.forEach(e=>{deg.set(e[0],(deg.get(e[0])||0)+1);deg.set(e[1],(deg.get(e[1])||0)+1);});
|
||||
nodes.sort((a,b)=>(deg.get(b.id)||0)-(deg.get(a.id)||0));
|
||||
const keep=new Set(nodes.slice(0,40).map(n=>n.id));nodes=nodes.filter(n=>keep.has(n.id));edges=edges.filter(e=>keep.has(e[0])&&keep.has(e[1]));
|
||||
}
|
||||
forceLayout(nodes,edges,W,H,300);
|
||||
let s=`<defs><filter id="glow2"><feGaussianBlur stdDeviation="5" result="blur"/><feMerge><feMergeNode in="blur"/><feMergeNode in="SourceGraphic"/></feMerge></filter></defs>`;
|
||||
for(const e of edges){const a=nodes.find(n=>n.id===e[0]),b=nodes.find(n=>n.id===e[1]);if(!a||!b)continue;
|
||||
s+=`<line class="g-edge" x1="${a.x.toFixed(1)}" y1="${a.y.toFixed(1)}" x2="${b.x.toFixed(1)}" y2="${b.y.toFixed(1)}"/>`;}
|
||||
for(const n of nodes){const col=NODE_COLOR[n.type]||NODE_COLOR.default,r=n.imp?13:8;
|
||||
s+=`<g class="g-node" transform="translate(${n.x.toFixed(1)},${n.y.toFixed(1)})"><circle r="${r}" fill="${col}"${n.imp?' filter="url(#glow2)"':""}/><text y="${r+13}">${esc(n.label||n.id).slice(0,18)}</text></g>`;}
|
||||
svg.innerHTML=s;
|
||||
}
|
||||
|
||||
function buildGraphLegend(types){
|
||||
let h="";
|
||||
for(const t of [...types].sort()){
|
||||
h+=`<span class="leg-item"><span class="leg-dot" style="background:${NODE_COLOR[t]||NODE_COLOR.default}"></span>${NODE_LABEL[t]||t}</span>`;
|
||||
}
|
||||
return h;
|
||||
}
|
||||
|
||||
/* ========== Simple markdown to HTML ========== */
|
||||
function mdToHtml(md){
|
||||
if(!md) return "";
|
||||
let html=esc(md);
|
||||
// code blocks
|
||||
html=html.replace(/```(\w*)\n([\s\S]*?)```/g,'<pre><code>$2</code></pre>');
|
||||
// inline code
|
||||
html=html.replace(/`([^`]+)`/g,'<code>$1</code>');
|
||||
// tables
|
||||
html=html.replace(/^(\|.+\|)\n(\|[-:\s|]+\|)\n((?:\|.+\|\n?)*)/gm,function(_,hdr,sep,body){
|
||||
const ths=hdr.split("|").filter(c=>c.trim()).map(c=>`<th>${c.trim()}</th>`).join("");
|
||||
let rows="";
|
||||
body.trim().split("\n").forEach(row=>{
|
||||
const tds=row.split("|").filter(c=>c.trim()).map(c=>`<td>${c.trim()}</td>`).join("");
|
||||
rows+=`<tr>${tds}</tr>`;
|
||||
});
|
||||
return `<table><thead><tr>${ths}</tr></thead><tbody>${rows}</tbody></table>`;
|
||||
});
|
||||
// headings
|
||||
html=html.replace(/^### (.+)$/gm,'<h3>$1</h3>');
|
||||
html=html.replace(/^## (.+)$/gm,'<h2>$1</h2>');
|
||||
html=html.replace(/^# (.+)$/gm,'<h1>$1</h1>');
|
||||
// bold
|
||||
html=html.replace(/\*\*([^*]+)\*\*/g,'<strong>$1</strong>');
|
||||
// unordered lists
|
||||
html=html.replace(/^- (.+)$/gm,'<li>$1</li>');
|
||||
html=html.replace(/(<li>.*<\/li>\n?)+/g,'<ul>$&</ul>');
|
||||
// paragraphs (lines not inside tags)
|
||||
html=html.replace(/^(?!<[huplo])((?!<).+)$/gm,'<p>$1</p>');
|
||||
return html;
|
||||
}
|
||||
|
||||
/* ========== MAIN RENDER ========== */
|
||||
async function loadResult(){
|
||||
const params=new URLSearchParams(window.location.search);
|
||||
const scenario=params.get("scenario");
|
||||
if(!scenario){
|
||||
document.getElementById("app").innerHTML='<div class="error-state"><h2>缺少场景参数</h2><p>请从仪表盘点击场景,或手动打开 <code>result.html?scenario=s1</code></p></div>';
|
||||
return;
|
||||
}
|
||||
|
||||
let data;
|
||||
try{
|
||||
const r=await fetch("/api/result/"+encodeURIComponent(scenario),{headers:{"X-Demo-Token":token()}});
|
||||
data=await r.json();
|
||||
}catch(e){
|
||||
document.getElementById("app").innerHTML='<div class="error-state"><h2>加载失败</h2><p>'+esc(e.message)+'</p><a class="back" href="index.html">返回仪表盘</a></div>';
|
||||
return;
|
||||
}
|
||||
if(!data.ok){
|
||||
document.getElementById("app").innerHTML='<div class="error-state"><h2>'+esc(data.error||"数据未就绪")+'</h2><p>请先在仪表盘运行该场景以生成结果。</p><a class="back" href="index.html">返回仪表盘</a></div>';
|
||||
return;
|
||||
}
|
||||
|
||||
const art=data.artifacts||{};
|
||||
const obj=art.json?JSON.parse(art.json):null;
|
||||
const owner=params.get("owner")||val_from("owner")||"";
|
||||
const repo=params.get("repo")||val_from("repo")||"";
|
||||
const keyword=params.get("keyword")||val_from("keyword")||"";
|
||||
|
||||
render(data,obj,art,scenario,owner,repo,keyword);
|
||||
}
|
||||
|
||||
function val_from(name){try{return localStorage.getItem("atlas_"+name)||"";}catch(e){return"";}}
|
||||
|
||||
function render(data,obj,art,scenario,owner,repo,keyword){
|
||||
const label=data.label||scenario;
|
||||
const desc=data.desc||"";
|
||||
const now=new Date().toLocaleString("zh-CN");
|
||||
let h="";
|
||||
|
||||
// ---- HEADER ----
|
||||
h+=`<div class="res-header">
|
||||
<a class="back no-print" href="index.html">← 返回仪表盘</a>
|
||||
<h1>${esc(label)}</h1>
|
||||
<div class="sub">${esc(desc)}</div>
|
||||
<div class="meta">`;
|
||||
if(owner||repo) h+=`<span class="chip">仓库 <b>${esc(owner)}${repo?"/"+esc(repo):""}</b></span>`;
|
||||
if(keyword) h+=`<span class="chip">关键词 <b>${esc(keyword)}</b></span>`;
|
||||
h+=`<span class="chip">场景 <b>${esc(scenario)}</b></span>
|
||||
<span class="chip">生成于 ${esc(now)}</span>
|
||||
</div>
|
||||
<div class="cmd-drawer no-print">
|
||||
<div class="dots"><i style="background:#FF5F57"></i><i style="background:#FEBC2E"></i><i style="background:#28C840"></i></div>
|
||||
<span style="color:var(--mint)">atlas ❯</span>
|
||||
<span class="cmd">gitlink-cli research --scenario ${esc(scenario)} ${owner?"--repo "+esc(owner)+"/"+esc(repo):""} ${keyword?'-k "'+esc(keyword)+'"':""}</span>
|
||||
</div></div>`;
|
||||
|
||||
// ---- METRICS ROW ----
|
||||
const metrics=metricsFor(scenario,obj);
|
||||
if(metrics){
|
||||
h+=`<div class="section"><div class="section-title"><span class="eyebrow">Key Metrics</span>关键指标</div>
|
||||
<div class="metrics-row">${metrics}</div></div>`;
|
||||
}
|
||||
|
||||
// ---- MAIN VIZ ----
|
||||
h+=`<div class="section"><div class="section-title"><span class="eyebrow">Visualization</span>可视化</div>`;
|
||||
|
||||
// S2: Knowledge Graph
|
||||
if(scenario==="s2"&&obj&&obj.nodes&&obj.edges){
|
||||
const graphNodes=obj.nodes.map(n=>({id:n.id,type:n.type||"default",label:n.label||n.id,imp:false}));
|
||||
// highlight repo and high-degree nodes
|
||||
const deg=new Map();
|
||||
obj.edges.forEach(e=>{deg.set((e.source||e.from),(deg.get(e.source||e.from)||0)+1);deg.set((e.target||e.to),(deg.get(e.target||e.to)||0)+1);});
|
||||
graphNodes.forEach(n=>{if(n.type==="repo"||deg.get(n.id)>3)n.imp=true;});
|
||||
const graphEdges=obj.edges.map(e=>[e.source||e.from,e.target||e.to]);
|
||||
const types=new Set(graphNodes.map(n=>n.type));
|
||||
const gId="resGraph_"+Date.now();
|
||||
h+=`<div class="graph-container">
|
||||
<div class="graph-svg-wrap"><svg id="${gId}" preserveAspectRatio="xMidYMid meet"></svg></div>
|
||||
<div class="graph-legend">${buildGraphLegend(types)}
|
||||
<span class="leg-note">${esc(label)} — ${(obj.meta||{}).repo_count||graphNodes.length} 个仓库节点, ${(obj.meta||{}).edge_count||graphEdges.length} 条关联边</span>
|
||||
</div></div>`;
|
||||
// defer drawing
|
||||
setTimeout(()=>drawGraphIn(gId,{nodes:graphNodes,edges:graphEdges}),100);
|
||||
}
|
||||
// S6: plotly
|
||||
else if(scenario==="s6"&&art.html){
|
||||
const blob=URL.createObjectURL(new Blob([art.html],{type:"text/html"}));
|
||||
h+=`<div class="viz-container"><iframe src="${blob}"></iframe></div>`;
|
||||
}
|
||||
// S1/S4: mermaid
|
||||
else if((scenario==="s1"||scenario==="s4")&&art.mermaid){
|
||||
const mId="resMmd_"+Date.now();
|
||||
const mmd=art.mermaid.replace(/```mermaid|```/g,"");
|
||||
h+=`<div class="mmd-container"><div class="mermaid" id="${mId}">${esc(mmd)}</div></div>`;
|
||||
setTimeout(()=>{const el=document.getElementById(mId);if(el)el.removeAttribute("data-processed");
|
||||
mermaid.run({nodes:["#"+mId]}).catch(()=>{});},100);
|
||||
}
|
||||
// S3: radar / scores
|
||||
else if(scenario==="s3"&&obj){
|
||||
const rs=obj.repro_score||0;const cs=obj.compliance_score||0;
|
||||
h+=`<div class="score-display">
|
||||
<div class="score-circle repro"><span class="val">${rs}</span><span class="label">复现性 / 10</span></div>
|
||||
<div class="score-circle compliance"><span class="val">${cs}</span><span class="label">合规性 / 10</span></div>
|
||||
</div>`;
|
||||
}
|
||||
// S5: weekly comparison + risks
|
||||
else if(scenario==="s5"&&obj){
|
||||
const ws=obj.week_stats||{};
|
||||
const tw=ws.this_week||{};const lw=ws.last_week||{};
|
||||
const maxC=Math.max(tw.commits||0,lw.commits||0,1);
|
||||
const maxP=Math.max(tw.prs_merged||0,lw.prs_merged||0,1);
|
||||
const maxI=Math.max(tw.issues_closed||0,lw.issues_closed||0,1);
|
||||
h+=`<div class="weekly-bars">
|
||||
<div style="font-weight:700;font-size:14px;margin-bottom:8px;color:var(--ink)">本周 vs 上周 对比</div>
|
||||
<div class="week-row"><span class="label">Commits</span><div class="bar-bg"><div class="bar-fill last-week" style="width:${(lw.commits/maxC*100||0).toFixed(1)}%"><span class="bar-val">${lw.commits||0}</span></div><div class="bar-fill this-week" style="width:${(tw.commits/maxC*100||0).toFixed(1)}%;position:relative;margin-top:-28px"><span class="bar-val">${tw.commits||0}</span></div></div></div>
|
||||
<div class="week-row"><span class="label">PR merged</span><div class="bar-bg"><div class="bar-fill last-week" style="width:${(lw.prs_merged/maxP*100||0).toFixed(1)}%"><span class="bar-val">${lw.prs_merged||0}</span></div><div class="bar-fill this-week" style="width:${(tw.prs_merged/maxP*100||0).toFixed(1)}%;position:relative;margin-top:-28px"><span class="bar-val">${tw.prs_merged||0}</span></div></div></div>
|
||||
<div class="week-row"><span class="label">Issues</span><div class="bar-bg"><div class="bar-fill last-week" style="width:${(lw.issues_closed/maxI*100||0).toFixed(1)}%"><span class="bar-val">${lw.issues_closed||0}</span></div><div class="bar-fill this-week" style="width:${(tw.issues_closed/maxI*100||0).toFixed(1)}%;position:relative;margin-top:-28px"><span class="bar-val">${tw.issues_closed||0}</span></div></div></div>
|
||||
</div>`;
|
||||
// risk warnings
|
||||
const risks=obj.risk_warnings||[];
|
||||
if(risks.length){
|
||||
h+=`<div style="margin-top:20px;font-weight:700;font-size:14px;color:var(--ink)">风险预警</div><div class="risk-list">`;
|
||||
risks.forEach(r=>{const cls=r.level==="critical"?"critical":"warning";
|
||||
h+=`<div class="risk-item ${cls}"><div class="type">${esc(r.level||"")} · ${esc(r.type||"")}</div><div class="msg">${esc(r.message||"")}</div>${r.suggestion?`<div class="suggestion">${esc(r.suggestion)}</div>`:""}</div>`;});
|
||||
h+=`</div>`;
|
||||
}
|
||||
}
|
||||
else{
|
||||
h+=`<div class="empty-state"><p>该场景无可视化产物,详见下方报告。</p></div>`;
|
||||
}
|
||||
h+=`</div>`;
|
||||
|
||||
// ---- REPORT ----
|
||||
if(art.report){
|
||||
h+=`<div class="section"><div class="section-title"><span class="eyebrow">Full Report</span>完整报告</div>
|
||||
<div class="report-content">${mdToHtml(art.report)}</div></div>`;
|
||||
}
|
||||
|
||||
// ---- STRUCTURED CONCLUSION CARDS ----
|
||||
const conclusions=buildConclusionCards(scenario,obj);
|
||||
if(conclusions){
|
||||
h+=`<div class="section"><div class="section-title"><span class="eyebrow">Conclusions</span>结构化结论</div>
|
||||
<div class="cards-grid">${conclusions}</div></div>`;
|
||||
}
|
||||
|
||||
document.getElementById("app").innerHTML=h;
|
||||
}
|
||||
|
||||
/* ========== METRICS ========== */
|
||||
function metricCard(label,val,note){
|
||||
return `<div class="metric-card"><div class="label">${esc(label)}</div><div class="value">${esc(String(val))}</div><div class="note">${esc(note||"")}</div></div>`;
|
||||
}
|
||||
function metricsFor(key,o){
|
||||
if(!o) return "";
|
||||
const M=o.meta||{};
|
||||
switch(key){
|
||||
case "s1": return metricCard("commits",M.commit_count||0,"采样提交")+metricCard("merged PR",M.merged_pr_count||0,"已合并")+metricCard("docs",M.doc_count||0,"文档文件")+metricCard("创新点",(o.innovation_points||[]).length,"高影响合并");
|
||||
case "s2": return metricCard("repos",M.repo_count||(o.nodes||[]).length,"仓库节点")+metricCard("scholars",M.scholar_count||0,"学者")+metricCard("topics",M.topic_count||0,"主题方向")+metricCard("edges",M.edge_count||(o.edges||[]).length,"关系边");
|
||||
case "s3": return metricCard("复现性",(o.repro_score||0)+"/10","reproducibility")+metricCard("合规性",(o.compliance_score||0)+"/10","compliance")+metricCard("风险项",(o.risks||o.risk_items||[]).length,"项")+metricCard("许可证",esc(licOf(o)),"识别");
|
||||
case "s4": return metricCard("缺口主题",(o.gap_topics||[]).length,"gap topics")+metricCard("推荐候选",(o.candidates||[]).length,"candidates")+metricCard("最高匹配",o.candidates&&o.candidates[0]?o.candidates[0].score:0,"top score")+metricCard("候选池",o.meta&&o.meta.pool_size||0,"pool size");
|
||||
case "s5":{const tw=(o.week_stats&&o.week_stats.this_week)||{};return metricCard("本周 commits",tw.commits||0,"")+metricCard("预警",(o.risk_warnings||[]).length,"risks")+metricCard("PR 合并",tw.prs_merged||0,"")+metricCard("活跃度",(o.trend&&o.trend.activity_level)||"—","");}
|
||||
case "s6":{const tl=o.timeline||{};const sm=a=>(a||[]).reduce((x,y)=>x+(y||0),0);return metricCard("周数",(tl.labels||[]).length,"")+metricCard("commits",sm(tl.commits),"")+metricCard("PRs",sm(tl.prs),"")+metricCard("issues",sm(tl.issues),"");}
|
||||
}
|
||||
return "";
|
||||
}
|
||||
|
||||
/* ========== CONCLUSION CARDS ========== */
|
||||
function buildConclusionCards(key,o){
|
||||
if(!o) return "";
|
||||
let cards="";
|
||||
switch(key){
|
||||
case "s1":{
|
||||
const ips=o.innovation_points||[];
|
||||
if(ips.length){
|
||||
cards+=`<div class="conclusion-card innovation"><div class="card-title">💡 创新点 (${ips.length})</div><div class="card-body">`;
|
||||
ips.slice(0,8).forEach(i=>{cards+=`<p><strong>[${esc(i.category)}]</strong> ${esc(i.description)}<br><small style="color:var(--mute)">${esc(i.evidence||"")}</small></p>`;});
|
||||
cards+=`</div></div>`;
|
||||
}
|
||||
break;}
|
||||
case "s2":{
|
||||
const scholars=o.core_scholars||[];
|
||||
if(scholars.length){
|
||||
cards+=`<div class="conclusion-card collab"><div class="card-title">👥 核心学者</div><div class="card-body">`;
|
||||
scholars.slice(0,6).forEach(s=>{cards+=`<p><strong>${esc(s.login)}</strong> — ${s.repo_count} 个仓库</p>`;});
|
||||
cards+=`</div></div>`;
|
||||
}
|
||||
const topics=o.topic_heat||[];
|
||||
if(topics.length){
|
||||
cards+=`<div class="conclusion-card innovation"><div class="card-title">🔥 热门主题</div><div class="card-body">`;
|
||||
topics.slice(0,6).forEach(t=>{cards+=`<p><strong>${esc(t.topic)}</strong> — ${t.count} 个仓库</p>`;});
|
||||
cards+=`</div></div>`;
|
||||
}
|
||||
break;}
|
||||
case "s3":{
|
||||
const risks=o.risk_items||o.risks||[];
|
||||
if(risks.length){
|
||||
cards+=`<div class="conclusion-card risk"><div class="card-title">⚠️ 风险项 (${risks.length})</div><div class="card-body">`;
|
||||
risks.slice(0,8).forEach(r=>{
|
||||
const lvl=(r.level||"").includes("high")||r.severity==="P0"?"🔴":"🟡";
|
||||
cards+=`<p>${lvl} <strong>${esc(r.category||r.name||"")}</strong> — ${esc(r.detail||r.evidence||"")}</p>`;});
|
||||
cards+=`</div></div>`;
|
||||
}
|
||||
break;}
|
||||
case "s4":{
|
||||
const cands=o.candidates||[];
|
||||
if(cands.length){
|
||||
cards+=`<div class="conclusion-card collab"><div class="card-title">🤝 协作推荐 (${cands.length})</div><div class="card-body">`;
|
||||
cands.slice(0,6).forEach((c,i)=>{cards+=`<p><strong>#${i+1} ${esc(c.login)}</strong> (${c.score}分)<br><small>${(c.reasons||[]).join(";")}</small></p>`;});
|
||||
cards+=`</div></div>`;
|
||||
}
|
||||
const gaps=o.gap_topics||[];
|
||||
if(gaps.length){
|
||||
cards+=`<div class="conclusion-card innovation"><div class="card-title">🔎 缺口主题 (${gaps.length})</div><div class="card-body">`;
|
||||
gaps.slice(0,6).forEach(g=>{cards+=`<p>${esc(typeof g==="string"?g:(g.topic||g.name||JSON.stringify(g)))}</p>`;});
|
||||
cards+=`</div></div>`;
|
||||
}
|
||||
break;}
|
||||
case "s5":{
|
||||
const risks=o.risk_warnings||[];
|
||||
if(risks.length){
|
||||
cards+=`<div class="conclusion-card risk"><div class="card-title">⚠️ 风险预警 (${risks.length})</div><div class="card-body">`;
|
||||
risks.slice(0,6).forEach(w=>{cards+=`<p><strong>[${esc(w.type)}]</strong> ${esc(w.message)}${w.suggestion?"<br><small>"+esc(w.suggestion)+"</small>":""}</p>`;});
|
||||
cards+=`</div></div>`;
|
||||
}
|
||||
const trend=o.trend||{};
|
||||
cards+=`<div class="conclusion-card innovation"><div class="card-title">📈 活跃趋势</div><div class="card-body"><p>commit 变化: <strong>${trend.commit_delta_pct||"—"}</strong></p><p>活跃度: <strong>${esc(trend.activity_level||"—")}</strong></p></div></div>`;
|
||||
break;}
|
||||
case "s6":{
|
||||
const tl=o.timeline||{};
|
||||
cards+=`<div class="conclusion-card innovation"><div class="card-title">📊 数据概览</div><div class="card-body"><p>时间线: ${(tl.labels||[]).length} 周</p><p>语言分布: ${(o.language_pie||[]).length} 类</p><p>累计 commits: ${(tl.commits||[]).reduce((a,b)=>a+(b||0),0)}</p></div></div>`;
|
||||
break;}
|
||||
}
|
||||
return cards;
|
||||
}
|
||||
|
||||
/* go */
|
||||
loadResult();
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
|
|
@ -0,0 +1,68 @@
|
|||
# Issue batch maintenance shortcuts
|
||||
|
||||
## Summary
|
||||
|
||||
Add OpenAPI-backed Issue batch maintenance shortcuts:
|
||||
|
||||
- `issue +batch-update` — batch update Issue status, priority, milestone, tags, and assigners by API issue IDs.
|
||||
- `issue +batch-delete` — batch delete Issues by API issue IDs with explicit confirmation.
|
||||
|
||||
This complements the existing `issue +batch-close` command. `batch-close` uses web URL issue numbers, while the OpenAPI batch update/delete endpoints use API issue IDs.
|
||||
|
||||
## OpenAPI coverage
|
||||
|
||||
| Command | Method | Endpoint |
|
||||
|---|---|---|
|
||||
| `issue +batch-update` | PATCH | `/api/v1/{owner}/{repo}/issues/batch_update.json` |
|
||||
| `issue +batch-delete` | DELETE | `/api/v1/{owner}/{repo}/issues/batch_destroy.json` |
|
||||
|
||||
## ID semantics
|
||||
|
||||
- `issue +batch-close --numbers` uses web URL Issue numbers (`project_issues_index`).
|
||||
- `issue +batch-update --ids` and `issue +batch-delete --ids` use API Issue IDs returned by Issue APIs.
|
||||
|
||||
The docs and help text explicitly call this out to avoid mixing the two ID types.
|
||||
|
||||
## Safety and usability
|
||||
|
||||
- Both commands support `--dry-run`.
|
||||
- `issue +batch-update` requires at least one update field.
|
||||
- `issue +batch-delete` is destructive and requires `--yes` for real execution.
|
||||
- ID lists are validated as positive integers and de-duplicated.
|
||||
|
||||
## Examples
|
||||
|
||||
```bash
|
||||
gitlink-cli issue +batch-update \
|
||||
--owner Gitlink \
|
||||
--repo forgeplus \
|
||||
--ids 101,102 \
|
||||
--status-id 3 \
|
||||
--priority-id 2 \
|
||||
--tag-ids 7,8 \
|
||||
--assigner-ids 11,12 \
|
||||
--dry-run
|
||||
|
||||
gitlink-cli issue +batch-delete \
|
||||
--owner Gitlink \
|
||||
--repo forgeplus \
|
||||
--ids 101,102 \
|
||||
--dry-run
|
||||
|
||||
gitlink-cli issue +batch-delete \
|
||||
--owner Gitlink \
|
||||
--repo forgeplus \
|
||||
--ids 101,102 \
|
||||
--yes
|
||||
```
|
||||
|
||||
## Tests
|
||||
|
||||
```bash
|
||||
GOPROXY=https://goproxy.cn,direct go test ./...
|
||||
go vet ./...
|
||||
go run . issue +batch-update --help
|
||||
go run . issue +batch-delete --help
|
||||
go run . issue +batch-update --owner wangyue111 --repo gitlink-cli --ids 101,102 --status-id 3 --dry-run --format json
|
||||
go run . issue +batch-delete --owner wangyue111 --repo gitlink-cli --ids 101,102 --dry-run --format json
|
||||
```
|
||||
|
|
@ -0,0 +1,23 @@
|
|||
# Wiki Shortcut
|
||||
|
||||
新增 `wiki` Shortcut 组,支持 Wiki 页面管理:
|
||||
|
||||
- `wiki +list` - 列出 Wiki 页面(目录结构)
|
||||
- `wiki +view` - 按页面名称查看 Wiki 页面详情
|
||||
- `wiki +create` - 创建新的 Wiki 页面
|
||||
- `wiki +update` - 更新 Wiki 页面标题和/或内容
|
||||
- `wiki +delete` - 删除 Wiki 页面
|
||||
|
||||
## 实现要点
|
||||
|
||||
- **API 端点**:基于 `/api/wiki/open/{action}` 扁平路径结构,覆盖 5 个 Wiki 管理接口:
|
||||
- `GET /api/wiki/open/wikiPages` — 目录列表
|
||||
- `GET /api/wiki/open/getWiki` — 查看页面
|
||||
- `POST /api/wiki/open/createWiki` — 创建页面
|
||||
- `PUT /api/wiki/open/updateWiki` — 更新页面
|
||||
- `DELETE /api/wiki/open/deleteWiki` — 删除页面
|
||||
- **标识方式**:Wiki 页面通过 `pageName`(slug)标识,所有操作需要 `projectId`(GitLink 项目数字 ID)
|
||||
- **内容编码**:创建和更新时,内容自动进行 base64 编码后以 `content_base64` 字段发送
|
||||
- **更新保护**:`+update` 要求必须提供 `--title` 和 `--page-name`;`--content` 为可选
|
||||
- **Shortcut 模式**:使用 `common.Shortcut` + `RuntimeContext` 框架,与其他模块保持一致
|
||||
|
||||
|
|
@ -43,6 +43,7 @@ func New() (*Client, error) {
|
|||
|
||||
// Do makes an API call with automatic .json suffix appended.
|
||||
func (c *Client) Do(method, path string, body interface{}, query url.Values) (*output.Envelope, error) {
|
||||
path = normalizeAPIPath(c.BaseURL, path)
|
||||
return c.do(method, path, body, query, true, "json")
|
||||
}
|
||||
|
||||
|
|
@ -62,10 +63,10 @@ func (c *Client) do(method, path string, body interface{}, query url.Values, app
|
|||
if idx := strings.Index(path, "?"); idx != -1 {
|
||||
basePath := path[:idx]
|
||||
queryStr := path[idx:]
|
||||
if !strings.HasSuffix(basePath, ".json") {
|
||||
if shouldAppendJSONSuffix(basePath) {
|
||||
path = basePath + ".json" + queryStr
|
||||
}
|
||||
} else if !strings.HasSuffix(path, ".json") {
|
||||
} else if shouldAppendJSONSuffix(path) {
|
||||
path += ".json"
|
||||
}
|
||||
}
|
||||
|
|
@ -144,22 +145,36 @@ func (c *Client) do(method, path string, body interface{}, query url.Values, app
|
|||
return output.SuccessEnvelope(string(respData), nil), nil
|
||||
}
|
||||
|
||||
// Check GitLink error-in-body pattern
|
||||
// Support both {"status":N, "message":"..."} and gateway {"code":N, "msg":"..."}
|
||||
var bodyCode float64
|
||||
var bodyMsg string
|
||||
if status, ok := raw["status"]; ok {
|
||||
var statusCode float64
|
||||
switch v := status.(type) {
|
||||
case float64:
|
||||
statusCode = v
|
||||
bodyCode = v
|
||||
case int:
|
||||
statusCode = float64(v)
|
||||
bodyCode = float64(v)
|
||||
}
|
||||
if statusCode != 0 && statusCode != 200 && statusCode != 1 {
|
||||
msg, _ := raw["message"].(string)
|
||||
suggestion := suggestFix(int(statusCode))
|
||||
return output.ErrorEnvelope(int(statusCode), msg, suggestion), &APIError{
|
||||
StatusCode: int(statusCode),
|
||||
Code: int(statusCode),
|
||||
Message: msg,
|
||||
}
|
||||
bodyMsg, _ = raw["message"].(string)
|
||||
} else if code, ok := raw["code"]; ok {
|
||||
switch v := code.(type) {
|
||||
case float64:
|
||||
bodyCode = v
|
||||
case int:
|
||||
bodyCode = float64(v)
|
||||
}
|
||||
bodyMsg, _ = raw["msg"].(string)
|
||||
if bodyMsg == "" {
|
||||
bodyMsg, _ = raw["message"].(string)
|
||||
}
|
||||
}
|
||||
if bodyCode != 0 && bodyCode != 200 && bodyCode != 201 && bodyCode != 204 && bodyCode != 1 {
|
||||
suggestion := suggestFix(int(bodyCode))
|
||||
return output.ErrorEnvelope(int(bodyCode), bodyMsg, suggestion), &APIError{
|
||||
StatusCode: int(bodyCode),
|
||||
Code: int(bodyCode),
|
||||
Message: bodyMsg,
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -197,6 +212,10 @@ func shouldAppendJSONSuffix(path string) bool {
|
|||
return false
|
||||
}
|
||||
}
|
||||
// Wiki open API endpoints do not use .json suffix
|
||||
if len(parts) >= 3 && parts[0] == "wiki" && parts[1] == "open" {
|
||||
return false
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -169,6 +169,45 @@ func TestClientDoStatusError(t *testing.T) {
|
|||
}
|
||||
}
|
||||
|
||||
func TestClientDoGatewayCodeError(t *testing.T) {
|
||||
// Gateway returns {"code":N, "msg":"..."} instead of {"status":N, "message":"..."}
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
w.Write([]byte(`{"code":400,"msg":"Bad Request"}`))
|
||||
}))
|
||||
defer server.Close()
|
||||
|
||||
c := &Client{HTTP: server.Client(), BaseURL: server.URL}
|
||||
env, err := c.Do("GET", "/api/test", nil, nil)
|
||||
if err == nil {
|
||||
t.Fatal("expected error for code=400")
|
||||
}
|
||||
if env == nil {
|
||||
t.Fatal("expected envelope for code error")
|
||||
}
|
||||
if env.OK {
|
||||
t.Fatal("expected OK=false for code=400")
|
||||
}
|
||||
}
|
||||
|
||||
func TestClientDoGatewayCode201Success(t *testing.T) {
|
||||
// Gateway returns code=201 with JSON string data — should be treated as success
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
w.Write([]byte(`{"code":201,"msg":"","data":"{\"title\":\"test\"}"}`))
|
||||
}))
|
||||
defer server.Close()
|
||||
|
||||
c := &Client{HTTP: server.Client(), BaseURL: server.URL}
|
||||
env, err := c.Do("POST", "/api/test", map[string]string{"title": "test"}, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected error for code=201: %v", err)
|
||||
}
|
||||
if !env.OK {
|
||||
t.Fatal("expected OK=true for code=201")
|
||||
}
|
||||
}
|
||||
|
||||
func TestClientDoStatusZero(t *testing.T) {
|
||||
// status=0, 200, 1 are treated as success
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
|
|
@ -525,3 +564,18 @@ func TestShouldAppendJSONSuffixSkipsExistingJSONPath(t *testing.T) {
|
|||
t.Fatal("existing .json path should not get another suffix")
|
||||
}
|
||||
}
|
||||
|
||||
func TestShouldAppendJSONSuffixSkipsWikiOpenPaths(t *testing.T) {
|
||||
paths := []string{
|
||||
"/wiki/open/createWiki",
|
||||
"/wiki/open/getWiki",
|
||||
"/wiki/open/updateWiki",
|
||||
"/wiki/open/deleteWiki",
|
||||
"/wiki/open/wikiPages",
|
||||
}
|
||||
for _, p := range paths {
|
||||
if shouldAppendJSONSuffix(p) {
|
||||
t.Errorf("wiki/open path %q should not get .json suffix", p)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -8,22 +8,25 @@ import (
|
|||
)
|
||||
|
||||
const (
|
||||
DefaultBaseURL = "https://www.gitlink.org.cn/api"
|
||||
DefaultFormat = "table"
|
||||
DefaultBaseURL = "https://www.gitlink.org.cn/api"
|
||||
DefaultGatewayURL = "https://gateway.gitlink.org.cn/api"
|
||||
DefaultFormat = "table"
|
||||
)
|
||||
|
||||
type Config struct {
|
||||
BaseURL string `yaml:"base_url"`
|
||||
Format string `yaml:"default_format"`
|
||||
Editor string `yaml:"editor,omitempty"`
|
||||
Pager string `yaml:"pager,omitempty"`
|
||||
Lang string `yaml:"lang,omitempty"`
|
||||
BaseURL string `yaml:"base_url"`
|
||||
GatewayURL string `yaml:"gateway_url"`
|
||||
Format string `yaml:"default_format"`
|
||||
Editor string `yaml:"editor,omitempty"`
|
||||
Pager string `yaml:"pager,omitempty"`
|
||||
Lang string `yaml:"lang,omitempty"`
|
||||
}
|
||||
|
||||
func DefaultConfig() *Config {
|
||||
return &Config{
|
||||
BaseURL: DefaultBaseURL,
|
||||
Format: DefaultFormat,
|
||||
BaseURL: DefaultBaseURL,
|
||||
GatewayURL: DefaultGatewayURL,
|
||||
Format: DefaultFormat,
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -54,6 +57,9 @@ func Load() (*Config, error) {
|
|||
if cfg.BaseURL == "" {
|
||||
cfg.BaseURL = DefaultBaseURL
|
||||
}
|
||||
if cfg.GatewayURL == "" {
|
||||
cfg.GatewayURL = DefaultGatewayURL
|
||||
}
|
||||
if cfg.Format == "" {
|
||||
cfg.Format = DefaultFormat
|
||||
}
|
||||
|
|
@ -80,6 +86,8 @@ func Get(key string) (string, error) {
|
|||
switch key {
|
||||
case "base_url":
|
||||
return cfg.BaseURL, nil
|
||||
case "gateway_url":
|
||||
return cfg.GatewayURL, nil
|
||||
case "default_format":
|
||||
return cfg.Format, nil
|
||||
case "editor":
|
||||
|
|
@ -101,6 +109,8 @@ func Set(key, value string) error {
|
|||
switch key {
|
||||
case "base_url":
|
||||
cfg.BaseURL = value
|
||||
case "gateway_url":
|
||||
cfg.GatewayURL = value
|
||||
case "default_format":
|
||||
cfg.Format = value
|
||||
case "editor":
|
||||
|
|
|
|||
|
|
@ -0,0 +1,59 @@
|
|||
# scripts/research — 子赛题四·科研辅助算法层
|
||||
|
||||
本目录是子赛题四「应用 GitLink 辅助科研」的 **Python 工具代码**(赛题交付物之一)。
|
||||
采用 **Go 出数据 + Python 做算法** 的分工:所有原始数据经现有 gitlink-cli(25 个域)获取,
|
||||
Python 负责知识图谱 / 协作匹配 / 可视化 / 复现性 / 报告等算法与产物生成。
|
||||
|
||||
## 文件
|
||||
|
||||
| 文件 | 场景 | 说明 |
|
||||
|------|------|------|
|
||||
| `gitlink_data.py` | 共享 | 调 gitlink-cli、解析 envelope、限速、分页、只读 health SQLite |
|
||||
| `collect.py` | 共享 | 各命令的采集器 + 字段归一化(login/repo 全名等) |
|
||||
| `lineage.py` | S1 | 仓库级科研项目洞悉:提交/分支/PR/文档/实验代码演进 + 创新点 |
|
||||
| `graph_build.py` | S2 | 科研知识图谱(networkx):节点/边模型 + 主题词典 + mermaid/DOT |
|
||||
| `repro.py` | S3 | 合规与复现性:license/密钥/依赖/复现清单 + 评分 |
|
||||
| `match.py` | S4 | 科研协作智能匹配:学者画像 × 仓库缺口 TF-IDF + 余弦 |
|
||||
| `report.py` | S5 | 科研进度智能跟踪与预警:周统计 + 里程碑 + 风险阈值 |
|
||||
| `visual.py` | S6 | 科研成果可视化:plotly 时间线/热力/饼/甘特/论文关联 |
|
||||
| `templates/*.j2` | 全部 | jinja2 中文报告模板 |
|
||||
| `test_*.py` | — | 单元测试(`pytest scripts/research/`) |
|
||||
|
||||
## 安装
|
||||
|
||||
```bash
|
||||
pip install -r scripts/research/requirements.txt
|
||||
```
|
||||
|
||||
## 环境变量
|
||||
|
||||
| 变量 | 默认 | 说明 |
|
||||
|------|------|------|
|
||||
| `GITLINK_CLI` | `gitlink-cli` | CLI 可执行路径;本地 Windows 开发可设 `./gitlink-cli.exe` |
|
||||
| `GITLINK_CLI_INTERVAL` | `0.6` | CLI 调用最小间隔(秒),防 API 限流 |
|
||||
| `GITLINK_HEALTH_DB` | `~/.agents/skills/gitlink-health/data/gitlink_health.db` | health SQLite 路径(图谱/匹配读历史协作数据) |
|
||||
|
||||
## 数据来源
|
||||
|
||||
- **结构化数据**:`repo/issue/pr/search/user/org/milestone/file/license` 等域,输出统一 envelope `{ok,data,meta}`。
|
||||
- **提交历史**:Raw API `gitlink-cli api GET /{owner}/{repo}/commits`(shortcuts 无 repo +commits,路径已在 mindspore 仓库核验通过)。
|
||||
- **历史协作**:直接只读 `health` SQLite(`users/repos/issues/pulls/tags` 表)。
|
||||
|
||||
## 验证仓库
|
||||
|
||||
**主验证仓库:`mindspore-Ecosystem/mindspore`**(华为 MindSpore 深度学习框架镜像,真实科研级 AI 框架):
|
||||
issues≈20346 / PR=9 / 贡献者=6 / star=31,协作数据充足,适合 S2/S4/S5;同时是 Python 研究代码仓库,适合 S1/S3/S6。
|
||||
S2 知识图谱为多仓库场景,按关键词 `search +repos` 跨仓库构建。
|
||||
|
||||
### 已核验响应形状(mindspore 实测,供各场景脚本参考)
|
||||
- `repo +info` → `default_branch`/`issues_count`/`pull_requests_count`/`contributor_users_count`/`size`/`clone_url`
|
||||
- `repo +contributors[]` → `login`/`name`/`contributions`/`contribution_perc`/`email`
|
||||
- `issue +list[]` → 标题 `subject`、时间 `created_at`、`status`、`priority_name`、`milestone_name`、`author`、`number`/`project_issues_index`
|
||||
- `pr +list[]` → 标题 `title`、创建时间 `pr_created_unix`(秒)、`status`(0=open/1=merged/2=closed)、`reviewers`、`index`
|
||||
- `api GET /commits[]` → `sha`/`message`/`timestamp`/`author.login`/`committer.login`
|
||||
- `repo +languages` → `{"Python":"99.7%", ...}`
|
||||
|
||||
## 复现
|
||||
|
||||
每个场景配一个 `skills/<skill>/examples/<scenario>-workflow.sh`,串联 CLI → Python → 报告产物,
|
||||
即赛题要求的「可复现执行脚本」。
|
||||
|
|
@ -0,0 +1,262 @@
|
|||
"""collect.py — 子赛题四共享数据采集器。
|
||||
|
||||
每个采集器是对 `gitlink-cli <domain> +<verb>` 或 Raw API 的薄封装,返回归一化后的
|
||||
Python 对象(dict/list)。所有 GitLink 操作经 gitlink-cli(见 gitlink-shared 工具边界)。
|
||||
|
||||
形状说明:GitLink 不同端点字段差异较大,本模块只做“尽力归一”,把不确定字段原样透传;
|
||||
具体字段解读留给各场景脚本(lineage/graph/match/...)并在其阶段用真实仓库验证后收敛。
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Iterable
|
||||
|
||||
import gitlink_data as gd
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 归一化小工具
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def as_str(v: Any) -> str:
|
||||
if v is None:
|
||||
return ""
|
||||
return str(v)
|
||||
|
||||
|
||||
def as_int(v: Any, default: int = 0) -> int:
|
||||
try:
|
||||
return int(v)
|
||||
except (TypeError, ValueError):
|
||||
return default
|
||||
|
||||
|
||||
def as_float(v: Any, default: float = 0.0) -> float:
|
||||
try:
|
||||
return float(v)
|
||||
except (TypeError, ValueError):
|
||||
return default
|
||||
|
||||
|
||||
def login_of(obj: dict) -> str:
|
||||
"""从用户/作者对象里尽量取出 login(GitLink 嵌套形式多变)。"""
|
||||
if not isinstance(obj, dict):
|
||||
return ""
|
||||
for path_keys in (("login",), ("name",), ("username",),
|
||||
("author", "login"), ("user", "login"),
|
||||
("owner", "login"), ("author", "name")):
|
||||
cur: Any = obj
|
||||
ok = True
|
||||
for k in path_keys:
|
||||
if isinstance(cur, dict) and k in cur:
|
||||
cur = cur[k]
|
||||
else:
|
||||
ok = False
|
||||
break
|
||||
if ok and isinstance(cur, str) and cur:
|
||||
return cur
|
||||
return ""
|
||||
|
||||
|
||||
def repo_fullname(project: dict) -> str:
|
||||
"""从 /projects 里的仓库对象取 'owner/identifier' 全名。"""
|
||||
if not isinstance(project, dict):
|
||||
return ""
|
||||
identifier = project.get("identifier") or project.get("name") or project.get("repo_name") or ""
|
||||
owner = login_of(project.get("author") or project.get("owner") or {}) or as_str(project.get("owner_login"))
|
||||
if owner and identifier:
|
||||
return f"{owner}/{identifier}"
|
||||
return identifier
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 仓库级
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def repo_info(owner: str, repo: str) -> dict:
|
||||
return gd.run_data(["repo", "+info"], owner=owner, repo=repo) or {}
|
||||
|
||||
|
||||
def readme(owner: str, repo: str, ref: str = "master") -> str:
|
||||
data = gd.run_data(["repo", "+readme", "--ref", ref], owner=owner, repo=repo)
|
||||
if isinstance(data, str):
|
||||
return data
|
||||
if isinstance(data, dict):
|
||||
for k in ("content", "text", "readme", "markdown", "data"):
|
||||
v = data.get(k)
|
||||
if isinstance(v, str) and v:
|
||||
return v
|
||||
return ""
|
||||
|
||||
|
||||
def tree(owner: str, repo: str, path: str = "", ref: str = "master") -> list:
|
||||
flags = ["repo", "+tree", "--ref", ref]
|
||||
if path:
|
||||
flags += ["--path", path]
|
||||
data = gd.run_data(flags, owner=owner, repo=repo)
|
||||
return gd.first_list(data, ("entries", "trees", "files", "sub_entries"))
|
||||
|
||||
|
||||
def languages(owner: str, repo: str) -> dict:
|
||||
data = gd.run_data(["repo", "+languages"], owner=owner, repo=repo)
|
||||
return data if isinstance(data, dict) else {}
|
||||
|
||||
|
||||
def contributors(owner: str, repo: str, limit: int = 100) -> list:
|
||||
"""贡献者列表。`repo +contributors` 无分页 flag(见 shortcuts/repo/repo.go),
|
||||
故单次取全量;limit 仅为兼容保留、不传给 CLI。"""
|
||||
data = gd.run_data(["repo", "+contributors"], owner=owner, repo=repo)
|
||||
return gd.first_list(data, ("contributors", "list"))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Issue / PR / 里程碑(分页)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def issues(owner: str, repo: str, state: str = "all", max_pages: int = 10,
|
||||
page_size: int = 50) -> list:
|
||||
extra = ["--state", state] if state and state != "all" else []
|
||||
return gd.paginate("issue", "list", owner=owner, repo=repo,
|
||||
page_size=page_size, max_pages=max_pages, extra_flags=extra)
|
||||
|
||||
|
||||
def prs(owner: str, repo: str, state: str = "all", max_pages: int = 10,
|
||||
page_size: int = 50) -> list:
|
||||
extra = ["--state", state] if state and state != "all" else []
|
||||
# GitLink 把 PR 列表也放在 data["issues"] 键下(见 shortcuts/health/api.go)。
|
||||
return gd.paginate("pr", "list", owner=owner, repo=repo,
|
||||
list_keys=("issues", "pulls", "list"),
|
||||
page_size=page_size, max_pages=max_pages, extra_flags=extra)
|
||||
|
||||
|
||||
def issues_all(owner: str, repo: str, max_pages: int = 10,
|
||||
page_size: int = 50) -> list:
|
||||
"""全部 Issue(open + closed 合并去重)。
|
||||
|
||||
`issue +list` 默认只返 open;取 open/closed 两个状态并集按 id 去重,
|
||||
每条保留真实 status 字段。供 S5/S6 做全量统计。
|
||||
"""
|
||||
seen: dict = {}
|
||||
for state in ("open", "closed"):
|
||||
for iss in issues(owner, repo, state=state,
|
||||
max_pages=max_pages, page_size=page_size):
|
||||
key = iss.get("id") or iss.get("index")
|
||||
if key is not None and key not in seen:
|
||||
seen[key] = iss
|
||||
return list(seen.values())
|
||||
|
||||
|
||||
def prs_all(owner: str, repo: str, max_pages: int = 10,
|
||||
page_size: int = 50) -> list:
|
||||
"""全部 PR(open + merged + closed 合并去重)。
|
||||
|
||||
GitLink 的 `pr +list` 状态过滤不可靠(不同仓库行为不一:有的默认只返 open,
|
||||
有的 --state 不生效返混合),故取三个状态并集按 index 去重,每条 PR 保留其
|
||||
真实 status 字段('merged'/'open'/'closed')供上层分类。仿 health 采集法。
|
||||
"""
|
||||
seen: dict = {}
|
||||
for state in ("open", "merged", "closed"):
|
||||
for pr in prs(owner, repo, state=state,
|
||||
max_pages=max_pages, page_size=page_size):
|
||||
key = pr.get("index") or pr.get("id") or pr.get("number")
|
||||
if key is None:
|
||||
continue
|
||||
if key not in seen:
|
||||
seen[key] = pr
|
||||
return list(seen.values())
|
||||
|
||||
|
||||
def pr_detail(owner: str, repo: str, index: int) -> dict:
|
||||
"""单个 PR 详情:`pr +view --id <index>`。
|
||||
|
||||
PR 列表 API 不返回文件改动数;详情接口返回 `files_count` / `commits_count`,
|
||||
用于 S1 创新点识别的「大规模重构」判据。
|
||||
"""
|
||||
return gd.run_data(["pr", "+view", "--id", str(index)], owner=owner, repo=repo) or {}
|
||||
|
||||
|
||||
def milestones(owner: str, repo: str, state: str = "all") -> list:
|
||||
extra = ["--status", state] if state and state != "all" else []
|
||||
data = gd.run_data(["milestone", "+list", *extra, "--limit", "100"],
|
||||
owner=owner, repo=repo)
|
||||
return gd.first_list(data, ("milestones", "list"))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 搜索 / 用户
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def search_repos(keyword: str, limit: int = 20) -> list:
|
||||
data = gd.run_data(["search", "+repos", "-k", keyword, "--limit", str(limit)])
|
||||
return gd.first_list(data, ("projects", "repos"))
|
||||
|
||||
|
||||
def search_users(keyword: str, limit: int = 20) -> list:
|
||||
data = gd.run_data(["search", "+users", "-k", keyword, "--limit", str(limit)])
|
||||
return gd.first_list(data, ("users", "list"))
|
||||
|
||||
|
||||
def user_info(login: str) -> dict:
|
||||
return gd.run_data(["user", "+info", "--login", login]) or {}
|
||||
|
||||
|
||||
def user_repos(login: str, limit: int = 20) -> list:
|
||||
"""某用户的公开仓库列表(`repo +list --user <login> --category all`)。
|
||||
|
||||
必须显式传 `--category all`:`repo +list` 的 category 默认是 `manage`
|
||||
("我管理的"),用来列别人的仓库时会 404/空。返回 data.projects,
|
||||
每项含 identifier(仓库名)/language({id,name})/description 等。
|
||||
"""
|
||||
data = gd.run_data(["repo", "+list", "--user", login,
|
||||
"--category", "all", "--limit", str(limit)])
|
||||
return gd.first_list(data, ("projects", "repos"))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 提交历史(Raw API:shortcuts 未提供 repo +commits)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def commits(owner: str, repo: str, ref: str = "master", max_pages: int = 5,
|
||||
page_size: int = 100) -> list:
|
||||
"""通过 `api GET /{owner}/{repo}/commits` 分页取提交。
|
||||
|
||||
注:该端点确切路径/分页参数需在 Phase 0c 用真实仓库核验;若不通,
|
||||
退化方案见 doc 注释(按分支取或用 compare 端点)。
|
||||
"""
|
||||
out: list = []
|
||||
for page in range(1, max_pages + 1):
|
||||
data = gd.api("GET", f"/{owner}/{repo}/commits",
|
||||
query=f"sha={ref}&page={page}&limit={page_size}",
|
||||
owner=owner, repo=repo)
|
||||
items = gd.first_list(data, ("commits", "list"))
|
||||
if not items:
|
||||
break
|
||||
out.extend(items)
|
||||
if len(items) < page_size:
|
||||
break
|
||||
return out
|
||||
|
||||
|
||||
def file_text(owner: str, repo: str, path: str, ref: str = "master") -> str:
|
||||
"""读取仓库内某文件的文本内容(LICENSE / CI 配置 / lockfile 等)。
|
||||
|
||||
`file +get`(sub_entries)返回形如 {"entries": {"content": "...", "commit": {...}}}。
|
||||
"""
|
||||
data = gd.run_data(["file", "+get", "--path", path, "--ref", ref],
|
||||
owner=owner, repo=repo)
|
||||
if isinstance(data, str):
|
||||
return data
|
||||
if isinstance(data, dict):
|
||||
# 顶层直接带内容
|
||||
for k in ("content", "text", "data"):
|
||||
v = data.get(k)
|
||||
if isinstance(v, str) and v:
|
||||
return v
|
||||
entries = data.get("entries")
|
||||
# 形式一:entries 是 dict,内含 content 键(GitLink 实测)
|
||||
if isinstance(entries, dict):
|
||||
v = entries.get("content") or entries.get("text")
|
||||
if isinstance(v, str) and v:
|
||||
return v
|
||||
# 形式二:entries 是 list[dict]
|
||||
elif isinstance(entries, list) and entries and isinstance(entries[0], dict):
|
||||
return as_str(entries[0].get("content") or entries[0].get("text"))
|
||||
return ""
|
||||
|
|
@ -0,0 +1,196 @@
|
|||
"""gitlink_data.py — 子赛题四数据访问共享层。
|
||||
|
||||
职责:
|
||||
1. 以子进程方式调用 gitlink-cli,解析统一 envelope({ok,data,error,meta})。
|
||||
2. 内置限速,避免触发 GitLink API 限流(参考 shortcuts/health 的 ~1.7 call/s)。
|
||||
3. 列表分页累积。
|
||||
4. 直接只读访问 health SQLite(shortcuts/health/schema.sql),供 S2 图谱 / S4 匹配取历史协作数据。
|
||||
|
||||
设计原则:本层只“取数 + 解析”,不做任何业务算法;算法在各场景脚本中实现。
|
||||
所有 GitLink 操作一律经 gitlink-cli,绝不用 gh/glab(见 skills/gitlink-shared/SKILL.md 工具边界)。
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import sqlite3
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Any, Iterable
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 配置
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# CLI 可执行路径:默认走 PATH 上的 gitlink-cli;本地 Windows 开发可设 GITLINK_CLI=./gitlink-cli.exe
|
||||
def cli_path() -> str:
|
||||
return os.environ.get("GITLINK_CLI", "gitlink-cli")
|
||||
|
||||
|
||||
# 最小调用间隔(秒),限制对 GitLink API 的请求频率。
|
||||
MIN_INTERVAL = float(os.environ.get("GITLINK_CLI_INTERVAL", "0.6"))
|
||||
|
||||
_last_call = 0.0
|
||||
|
||||
|
||||
def _throttle() -> None:
|
||||
"""简单的全局令牌限速:两次调用间至少间隔 MIN_INTERVAL 秒。"""
|
||||
global _last_call
|
||||
now = time.time()
|
||||
wait = MIN_INTERVAL - (now - _last_call)
|
||||
if wait > 0:
|
||||
time.sleep(wait)
|
||||
_last_call = time.time()
|
||||
|
||||
|
||||
def _warn(msg: str) -> None:
|
||||
sys.stderr.write(f"[gitlink] {msg}\n")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 核心:调用 CLI
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def run(args: Iterable[str], owner: str | None = None, repo: str | None = None,
|
||||
fmt: str = "json") -> dict[str, Any]:
|
||||
"""调用 `gitlink-cli [global flags] args... --format fmt`,返回解析后的 envelope dict。
|
||||
|
||||
失败时返回 {"ok": False, "error": {...}},不抛异常,便于调用方容错。
|
||||
"""
|
||||
cmd = [cli_path()]
|
||||
if owner:
|
||||
cmd += ["--owner", owner]
|
||||
if repo:
|
||||
cmd += ["--repo", repo]
|
||||
cmd += list(args)
|
||||
if fmt:
|
||||
cmd += ["--format", fmt]
|
||||
|
||||
_throttle()
|
||||
try:
|
||||
proc = subprocess.run(cmd, capture_output=True, text=True, timeout=120)
|
||||
except FileNotFoundError:
|
||||
_warn(f"gitlink-cli 未找到(cmd={cmd[0]}),请设置 GITLINK_CLI 环境变量")
|
||||
return {"ok": False, "error": {"message": f"gitlink-cli not found: {cmd[0]}"}}
|
||||
except subprocess.TimeoutExpired:
|
||||
_warn(f"调用超时:{' '.join(cmd)}")
|
||||
return {"ok": False, "error": {"message": "timeout"}}
|
||||
|
||||
if proc.returncode != 0:
|
||||
_warn(f"exit={proc.returncode} cmd={' '.join(cmd)}\n{proc.stderr.strip()}")
|
||||
return {"ok": False, "error": {"message": (proc.stderr or proc.stdout).strip() or f"exit {proc.returncode}"}}
|
||||
|
||||
out = proc.stdout.strip()
|
||||
if not out:
|
||||
return {"ok": True, "data": None}
|
||||
try:
|
||||
return json.loads(out)
|
||||
except json.JSONDecodeError:
|
||||
# 非 JSON(表格/原始文本),原样包裹返回
|
||||
return {"ok": True, "data": out}
|
||||
|
||||
|
||||
def run_data(args: Iterable[str], owner: str | None = None, repo: str | None = None,
|
||||
fmt: str = "json") -> Any:
|
||||
"""run() 的便捷封装:成功返回 data 字段,失败返回 None 并告警。"""
|
||||
env = run(args, owner=owner, repo=repo, fmt=fmt)
|
||||
if not env.get("ok"):
|
||||
msg = (env.get("error") or {}).get("message", "unknown error")
|
||||
_warn(f"{list(args)} -> {msg}")
|
||||
return None
|
||||
return env.get("data")
|
||||
|
||||
|
||||
def api(method: str, path: str, query: str | None = None,
|
||||
owner: str | None = None, repo: str | None = None) -> Any:
|
||||
"""调用 Raw API:`gitlink-cli api METHOD PATH --query ... --format json`。
|
||||
|
||||
用于 shortcuts 未覆盖的端点(如仓库提交历史 GET /{owner}/{repo}/commits)。
|
||||
"""
|
||||
args = ["api", method, path]
|
||||
if query:
|
||||
args += ["--query", query]
|
||||
return run_data(args, owner=owner, repo=repo)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 分页 / 形状工具
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# GitLink 各端点把列表放在不同键下;按优先级尝试这些键。
|
||||
DEFAULT_LIST_KEYS = (
|
||||
"projects", "repos", "issues", "pulls", "users", "list",
|
||||
"contributors", "entries", "commits", "milestones", "tags",
|
||||
)
|
||||
|
||||
|
||||
def first_list(data: Any, keys: Iterable[str] = DEFAULT_LIST_KEYS) -> list:
|
||||
"""从 envelope.data 里稳健地取出列表:data 本身是列表则直接返回,否则尝试已知键。"""
|
||||
if isinstance(data, list):
|
||||
return data
|
||||
if isinstance(data, dict):
|
||||
for k in keys:
|
||||
v = data.get(k)
|
||||
if isinstance(v, list):
|
||||
return v
|
||||
# 兜底:唯一一个 list 值
|
||||
list_vals = [v for v in data.values() if isinstance(v, list)]
|
||||
if len(list_vals) == 1:
|
||||
return list_vals[0]
|
||||
return []
|
||||
|
||||
|
||||
def total_count(data: Any) -> int | None:
|
||||
if isinstance(data, dict):
|
||||
for k in ("total_count", "totalCount", "count", "total"):
|
||||
if isinstance(data.get(k), (int, float)):
|
||||
return int(data[k])
|
||||
return None
|
||||
|
||||
|
||||
def paginate(domain: str, verb: str, list_keys=DEFAULT_LIST_KEYS,
|
||||
page_size: int = 50, max_pages: int = 20,
|
||||
owner: str | None = None, repo: str | None = None,
|
||||
extra_flags: Iterable[str] = ()) -> list:
|
||||
"""对一个 `gitlink-cli <domain> +<verb>` 列表命令做多页累积。
|
||||
|
||||
依赖命令支持 --page/--limit 两个 flag(repo/issue/pr/search/milestone 等均支持)。
|
||||
"""
|
||||
collected: list = []
|
||||
for page in range(1, max_pages + 1):
|
||||
flags = [f"+{verb}", "--page", str(page), "--limit", str(page_size), *extra_flags]
|
||||
data = run_data([domain, *flags], owner=owner, repo=repo)
|
||||
if data is None:
|
||||
break
|
||||
items = first_list(data, list_keys)
|
||||
if not items:
|
||||
break
|
||||
collected.extend(items)
|
||||
total = total_count(data)
|
||||
if total is not None and len(collected) >= total:
|
||||
break
|
||||
if len(items) < page_size:
|
||||
break
|
||||
return collected
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# health SQLite 只读访问
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def health_db_path() -> str:
|
||||
return os.environ.get(
|
||||
"GITLINK_HEALTH_DB",
|
||||
str(Path.home() / ".agents" / "skills" / "gitlink-health" / "data" / "gitlink_health.db"),
|
||||
)
|
||||
|
||||
|
||||
def open_health_db() -> sqlite3.Connection | None:
|
||||
"""以只读方式打开 health SQLite;文件不存在则返回 None(调用方退化为纯 API 取数)。"""
|
||||
p = health_db_path()
|
||||
if not Path(p).exists():
|
||||
return None
|
||||
# mode=ro 防止误写;modernc/sqlite 已开 WAL,Python 只读并发安全。
|
||||
return sqlite3.connect(f"file:{p}?mode=ro", uri=True)
|
||||
|
|
@ -0,0 +1,658 @@
|
|||
"""graph_build.py — S2 科研热点追踪与知识图谱。
|
||||
|
||||
输入一组科研关键词,从 GitLink 平台按关键词搜索相关仓库(search +repos),
|
||||
对每个候选仓库取 repo_info / contributors / languages / README,再用
|
||||
networkx.MultiDiGraph 构建一张「仓库—学者—主题」科研知识图谱:
|
||||
|
||||
节点
|
||||
- repo : id = repo:owner/name (props: language/stars/forks/desc)
|
||||
- scholar : id = scholar:login (来自 contributors,过滤 bot/i-robot)
|
||||
- topic : id = topic:x (由 topics.py 词典抽取)
|
||||
边
|
||||
- contributes_to : scholar → repo (weight = contribution_perc 解析为 0~1)
|
||||
- owns : scholar → repo (当 author.login == contributor login)
|
||||
- covers_topic : repo → topic (weight = 出现次数 / max)
|
||||
- collaborates_with : scholar → scholar (共享同一 repo)
|
||||
- related_to : topic ↔ topic (在同一 repo 共现)
|
||||
|
||||
「取数」与「建图」严格分离:build_graph() 只接收已经取好的 Python 数据结构,
|
||||
便于离线单测(不联网、不调 gitlink-cli)。collect() 负责在线取数。
|
||||
|
||||
数据全部经 gitlink-cli 获取(search +repos / repo +info / repo +contributors /
|
||||
repo +languages / repo +readme)。
|
||||
|
||||
用法:
|
||||
python graph_build.py --keywords "deep learning,nlp" --repos-limit 20 --out ./out
|
||||
python graph_build.py --keywords "knowledge graph" # 仅打印 JSON
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import datetime as _dt
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
from collections import Counter, defaultdict
|
||||
from typing import Any
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
import collect as c # noqa: E402
|
||||
import topics as T # noqa: E402
|
||||
|
||||
import networkx as nx # noqa: E402
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 小工具
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
BOT_LOGIN_HINTS = ("bot", "i-robot", "dependabot", "renovate", "semantic-release-bot")
|
||||
|
||||
|
||||
def is_bot(login: str) -> bool:
|
||||
"""识别明显机器人账号(不作为学者节点)。"""
|
||||
if not login:
|
||||
return True
|
||||
low = login.lower()
|
||||
return any(h in low for h in BOT_LOGIN_HINTS)
|
||||
|
||||
|
||||
def parse_ratio(v: Any) -> float:
|
||||
"""把 '1.18%' / '0.2' / 0.2 等统一解析为 0~1 比例(与 match.py 一致)。"""
|
||||
if v is None:
|
||||
return 0.0
|
||||
s = str(v).strip()
|
||||
pct = s.endswith("%")
|
||||
if pct:
|
||||
s = s[:-1]
|
||||
try:
|
||||
f = float(s)
|
||||
except ValueError:
|
||||
return 0.0
|
||||
return f / 100.0 if (pct or f > 1.0) else f
|
||||
|
||||
|
||||
def _topic_heat(descriptions: list[str], top: int = 10) -> list[dict]:
|
||||
"""对全部仓库 description 跑 topics.topic_counter,取 top 热度榜。"""
|
||||
cnt = T.topic_counter(descriptions)
|
||||
return [{"topic": t, "count": n} for t, n in cnt.most_common(top)]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 热点追踪:飙升项目 + 活跃讨论(快照代理;真·增长率需定时轮询存历史)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _to_epoch(v: Any) -> float:
|
||||
"""把 GitLink 时间(ISO 字符串或整数秒)解析为 epoch 秒,失败返回 0.0。"""
|
||||
if v is None:
|
||||
return 0.0
|
||||
if isinstance(v, (int, float)):
|
||||
return float(v) / 1000.0 if v > 1e12 else float(v)
|
||||
s = str(v).strip()
|
||||
if not s:
|
||||
return 0.0
|
||||
if s.isdigit():
|
||||
f = float(s)
|
||||
return f / 1000.0 if f > 1e12 else f
|
||||
iso = s.replace("Z", "+00:00")
|
||||
try:
|
||||
return _dt.datetime.fromisoformat(iso).timestamp()
|
||||
except (ValueError, TypeError):
|
||||
import re
|
||||
m = re.search(r"(\d{4})-(\d{2})-(\d{2})", s)
|
||||
if m:
|
||||
try:
|
||||
return _dt.datetime(int(m.group(1)), int(m.group(2)), int(m.group(3))).timestamp()
|
||||
except ValueError:
|
||||
return 0.0
|
||||
return 0.0
|
||||
|
||||
|
||||
def _iso_day(epoch: float) -> str:
|
||||
if not epoch:
|
||||
return ""
|
||||
try:
|
||||
return _dt.datetime.utcfromtimestamp(epoch).strftime("%Y-%m-%d")
|
||||
except (OSError, ValueError, OverflowError):
|
||||
return ""
|
||||
|
||||
|
||||
def compute_trending(repos: list[dict], top: int = 15) -> list[dict]:
|
||||
"""对搜索到的仓库按「热度分数」排序,作为飙升/热门项目代理。
|
||||
|
||||
分数 = star + fork×2 + 近期更新加成;并给「日均增星(velocity)」作为
|
||||
day-1 可用的趋势代理(真·增长率需定时轮询存历史快照,见 trends 表规划)。
|
||||
"""
|
||||
now = time.time()
|
||||
scored: list[dict] = []
|
||||
for r in repos:
|
||||
fullname = r.get("fullname") or c.repo_fullname(r)
|
||||
if not fullname:
|
||||
continue
|
||||
stars = c.as_int(r.get("praises_count"))
|
||||
forks = c.as_int(r.get("forked_count"))
|
||||
updated = _to_epoch(r.get("updated_at") or r.get("time")
|
||||
or r.get("updated_on") or r.get("created_at"))
|
||||
age_days = max(1.0, (now - updated) / 86400.0) if updated else 99999.0
|
||||
recency = max(0.0, 60.0 - age_days) # 60 天内更新有加成
|
||||
score = stars + forks * 2 + recency
|
||||
velocity = round(stars / age_days, 3) if age_days < 99990 else 0.0
|
||||
lang_obj = r.get("language")
|
||||
lang = lang_obj.get("name") if isinstance(lang_obj, dict) else c.as_str(lang_obj)
|
||||
scored.append({
|
||||
"repo": fullname,
|
||||
"description": c.as_str(r.get("description"))[:140],
|
||||
"language": lang or "",
|
||||
"stars": stars,
|
||||
"forks": forks,
|
||||
"updated": _iso_day(updated),
|
||||
"velocity": velocity,
|
||||
"score": round(score, 1),
|
||||
})
|
||||
scored.sort(key=lambda x: -x["score"])
|
||||
return scored[:top]
|
||||
|
||||
|
||||
def compute_active(issues_map: dict[str, list], prs_map: dict[str, list],
|
||||
top: int = 12) -> list[dict]:
|
||||
"""跨仓库取评论/日志最多的 Issue / PR,作为「活跃讨论」信号。"""
|
||||
items: list[dict] = []
|
||||
for fullname, issues in issues_map.items():
|
||||
for iss in issues or []:
|
||||
if not isinstance(iss, dict):
|
||||
continue
|
||||
jc = c.as_int(iss.get("journals_count") or iss.get("comments_count"))
|
||||
items.append({
|
||||
"repo": fullname, "type": "issue",
|
||||
"number": iss.get("index") or iss.get("number") or iss.get("id"),
|
||||
"title": c.as_str(iss.get("subject") or iss.get("title"))[:120],
|
||||
"comments": jc,
|
||||
"state": c.as_str(iss.get("status") or iss.get("issue_status") or "open"),
|
||||
})
|
||||
for fullname, prs in prs_map.items():
|
||||
for pr in prs or []:
|
||||
if not isinstance(pr, dict):
|
||||
continue
|
||||
jc = c.as_int(pr.get("journals_count") or pr.get("comments_count"))
|
||||
items.append({
|
||||
"repo": fullname, "type": "pr",
|
||||
"number": pr.get("index") or pr.get("number") or pr.get("id"),
|
||||
"title": c.as_str(pr.get("title"))[:120],
|
||||
"comments": jc,
|
||||
"state": c.as_str(pr.get("status") or "open"),
|
||||
})
|
||||
items.sort(key=lambda x: -x["comments"])
|
||||
# 优先有评论的;不足则按已有顺序补齐
|
||||
commented = [it for it in items if it["comments"] > 0]
|
||||
return (commented or items)[:top]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 建图(纯函数:不联网,只吃已取数据,便于单测)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def build_graph(repos: list[dict],
|
||||
contributors_map: dict[str, list[dict]],
|
||||
languages_map: dict[str, dict],
|
||||
readmes: dict[str, str],
|
||||
keywords: list[str] | None = None) -> dict[str, Any]:
|
||||
"""构建科研知识图谱。
|
||||
|
||||
参数(全部为「已取好」的 Python 数据,不联网):
|
||||
repos : list[dict],每个元素至少含 fullname 与 repo_info 字段
|
||||
(identifier, author.login, description, language.name,
|
||||
praises_count, forked_count)。
|
||||
contributors_map : {fullname: [contributor, ...]},contributor 至少含
|
||||
login + contribution_perc。
|
||||
languages_map : {fullname: {"Python": "99.7%", ...}}。
|
||||
readmes : {fullname: readme 文本(已截断到前 4000 字符)}。
|
||||
|
||||
返回结果 dict(与 graph.json 结构一致)。
|
||||
"""
|
||||
G = nx.MultiDiGraph()
|
||||
|
||||
repo_nodes: dict[str, dict] = {}
|
||||
scholar_repos: dict[str, set[str]] = defaultdict(set)
|
||||
repo_topics: dict[str, Counter] = {}
|
||||
descriptions: list[str] = []
|
||||
max_topic_count = 1 # 用于 covers_topic 归一化(防除零)
|
||||
|
||||
# ---- 1) 仓库 + 主题节点 ----
|
||||
for r in repos:
|
||||
fullname = r.get("fullname") or c.repo_fullname(r)
|
||||
if not fullname:
|
||||
continue
|
||||
desc = c.as_str(r.get("description"))
|
||||
lang_name = ""
|
||||
lang_obj = r.get("language")
|
||||
if isinstance(lang_obj, dict):
|
||||
lang_name = c.as_str(lang_obj.get("name"))
|
||||
stars = c.as_int(r.get("praises_count"))
|
||||
forks = c.as_int(r.get("forked_count"))
|
||||
readme = c.as_str(readmes.get(fullname))[:4000]
|
||||
descriptions.append(desc)
|
||||
|
||||
repo_id = f"repo:{fullname}"
|
||||
props = {
|
||||
"language": lang_name,
|
||||
"stars": stars,
|
||||
"forks": forks,
|
||||
"description": desc,
|
||||
"readme_head": readme[:200],
|
||||
}
|
||||
G.add_node(repo_id, type="repo", label=fullname, **props)
|
||||
repo_nodes[repo_id] = {"label": fullname, **props}
|
||||
|
||||
# 抽取主题(description + readme)
|
||||
tps = T.extract_topics(desc + " " + readme)
|
||||
cnt: Counter = Counter()
|
||||
for tp in tps:
|
||||
cnt[tp] += 1
|
||||
repo_topics[repo_id] = cnt
|
||||
if cnt:
|
||||
max_topic_count = max(max_topic_count, max(cnt.values()))
|
||||
|
||||
# ---- 2) 主题节点 + covers_topic 边 ----
|
||||
topic_repos: dict[str, set[str]] = defaultdict(set)
|
||||
repo_topic_pairs: dict[str, list[str]] = {} # repo_id -> [topic_id]
|
||||
for repo_id, cnt in repo_topics.items():
|
||||
pairs: list[str] = []
|
||||
for tp, n in cnt.items():
|
||||
topic_id = f"topic:{tp}"
|
||||
if topic_id not in G:
|
||||
G.add_node(topic_id, type="topic", label=tp, count=0,
|
||||
language="", stars=0, forks=0, description="")
|
||||
# 累计该主题被多少仓库覆盖
|
||||
G.nodes[topic_id]["count"] += 1
|
||||
weight = n / max_topic_count if max_topic_count else 0.0
|
||||
G.add_edge(repo_id, topic_id, type="covers_topic",
|
||||
weight=round(weight, 4))
|
||||
topic_repos[tp].add(repo_id)
|
||||
pairs.append(topic_id)
|
||||
repo_topic_pairs[repo_id] = pairs
|
||||
|
||||
# ---- 3) 学者节点 + contributes_to / owns ----
|
||||
for fullname, contribs in contributors_map.items():
|
||||
repo_id = f"repo:{fullname}"
|
||||
if repo_id not in G:
|
||||
continue
|
||||
owner_login = ""
|
||||
# 从 repos 列表里取该仓库 author.login(判定 owns)
|
||||
for r in repos:
|
||||
if (r.get("fullname") or c.repo_fullname(r)) == fullname:
|
||||
owner_login = c.login_of(r.get("author") or {})
|
||||
break
|
||||
for contrib in contribs:
|
||||
login = c.login_of(contrib)
|
||||
if is_bot(login):
|
||||
continue
|
||||
scholar_id = f"scholar:{login}"
|
||||
if scholar_id not in G:
|
||||
G.add_node(scholar_id, type="scholar", label=login,
|
||||
language="", stars=0, forks=0, description="")
|
||||
weight = parse_ratio(contrib.get("contribution_perc"))
|
||||
G.add_edge(scholar_id, repo_id, type="contributes_to",
|
||||
weight=round(weight, 4))
|
||||
if login == owner_login:
|
||||
G.add_edge(scholar_id, repo_id, type="owns", weight=1.0)
|
||||
scholar_repos[login].add(repo_id)
|
||||
|
||||
# ---- 4) collaborates_with(共享同一 repo 的两两学者)----
|
||||
for contribs in contributors_map.values():
|
||||
logins = [c.login_of(x) for x in contribs if not is_bot(c.login_of(x))]
|
||||
logins = sorted(set(logins))
|
||||
if len(logins) < 2:
|
||||
continue
|
||||
for i in range(len(logins)):
|
||||
for j in range(i + 1, len(logins)):
|
||||
a = f"scholar:{logins[i]}"
|
||||
b = f"scholar:{logins[j]}"
|
||||
# 双向(无向语义;MultiDiGraph 用两条边近似)
|
||||
G.add_edge(a, b, type="collaborates_with", weight=1.0)
|
||||
G.add_edge(b, a, type="collaborates_with", weight=1.0)
|
||||
|
||||
# ---- 5) related_to(同一 repo 内共现的两两主题)----
|
||||
for repo_id, tids in repo_topic_pairs.items():
|
||||
for i in range(len(tids)):
|
||||
for j in range(i + 1, len(tids)):
|
||||
a, b = tids[i], tids[j]
|
||||
G.add_edge(a, b, type="related_to", weight=1.0)
|
||||
G.add_edge(b, a, type="related_to", weight=1.0)
|
||||
|
||||
# ---- 6) 导出 ----
|
||||
nodes_out = []
|
||||
for nid, attrs in G.nodes(data=True):
|
||||
nodes_out.append({
|
||||
"id": nid,
|
||||
"type": attrs.get("type", ""),
|
||||
"label": attrs.get("label", nid),
|
||||
"props": {k: v for k, v in attrs.items()
|
||||
if k not in ("type", "label")},
|
||||
})
|
||||
edges_out = []
|
||||
for u, v, attrs in G.edges(data=True):
|
||||
edges_out.append({
|
||||
"source": u,
|
||||
"target": v,
|
||||
"type": attrs.get("type", ""),
|
||||
"weight": attrs.get("weight", 1.0),
|
||||
})
|
||||
|
||||
# 核心学者:按出现 repo 数排序
|
||||
core_scholars = sorted(
|
||||
({"login": lg, "repo_count": len(rs)} for lg, rs in scholar_repos.items()),
|
||||
key=lambda x: (-x["repo_count"], x["login"]),
|
||||
)[:15]
|
||||
|
||||
# 核心团队:仅统计「组织」类型(非个人 User)的仓库拥有者,按拥有仓库数排序。
|
||||
# (个人账号不算团队;author.type 区分 User / Organization)
|
||||
owner_count: dict[str, int] = {}
|
||||
owner_is_org: dict[str, bool] = {}
|
||||
for r in repos:
|
||||
author = r.get("author") or {}
|
||||
lg = c.login_of(author)
|
||||
if not lg:
|
||||
continue
|
||||
owner_count[lg] = owner_count.get(lg, 0) + 1
|
||||
tp = str(author.get("type", "")).lower()
|
||||
if tp and tp not in ("user", ""):
|
||||
owner_is_org[lg] = True
|
||||
owner_logins = sorted(
|
||||
[{"login": lg, "repo_count": cnt, "type": "organization"}
|
||||
for lg, cnt in owner_count.items() if owner_is_org.get(lg)],
|
||||
key=lambda x: (-x["repo_count"], x["login"]),
|
||||
)
|
||||
|
||||
# 主题热度:取所有 description 的 top10(含图谱里实际命中的 count)
|
||||
heat = _topic_heat(descriptions, top=10)
|
||||
|
||||
return {
|
||||
"scenario": "S2_research_knowledge_graph",
|
||||
"keywords": list(keywords or []),
|
||||
"nodes": nodes_out,
|
||||
"edges": edges_out,
|
||||
"core_scholars": core_scholars,
|
||||
"core_teams": owner_logins,
|
||||
"topic_heat": heat,
|
||||
"meta": {
|
||||
"keywords": list(keywords or []),
|
||||
"repo_count": len(repo_nodes),
|
||||
"node_count": G.number_of_nodes(),
|
||||
"edge_count": G.number_of_edges(),
|
||||
"scholar_count": sum(1 for n in nodes_out if n["type"] == "scholar"),
|
||||
"topic_count": sum(1 for n in nodes_out if n["type"] == "topic"),
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 取数(在线:调 gitlink-cli)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def collect(keywords: list[str], repos_limit: int = 20) -> dict[str, Any]:
|
||||
"""按关键词搜索仓库并取其 info/contributors/languages/readme,返回原始数据。
|
||||
|
||||
与 build_graph() 解耦:本函数可被替换为 mock(单测里直接构造数据喂 build_graph)。
|
||||
"""
|
||||
seen: dict[str, dict] = {} # fullname -> 归一化的 repo dict
|
||||
for kw in keywords:
|
||||
for r in c.search_repos(kw, limit=repos_limit):
|
||||
fullname = c.repo_fullname(r)
|
||||
if not fullname or fullname in seen:
|
||||
continue
|
||||
seen[fullname] = _normalize_search_hit(r, fullname)
|
||||
if len(seen) >= repos_limit:
|
||||
break
|
||||
|
||||
repos = list(seen.values())[:repos_limit]
|
||||
contributors_map: dict[str, list[dict]] = {}
|
||||
languages_map: dict[str, dict] = {}
|
||||
readmes: dict[str, str] = {}
|
||||
issues_map: dict[str, list[dict]] = {}
|
||||
prs_map: dict[str, list[dict]] = {}
|
||||
for r in repos:
|
||||
fullname = r["fullname"]
|
||||
owner, _, name = fullname.partition("/")
|
||||
# 用仓库完整 info 覆盖搜索结果的稀疏字段
|
||||
info = c.repo_info(owner, name)
|
||||
if info:
|
||||
r["description"] = c.as_str(info.get("description")) or r.get("description", "")
|
||||
r["praises_count"] = c.as_int(info.get("praises_count") or info.get("watchers_count"))
|
||||
r["forked_count"] = c.as_int(info.get("forked_count"))
|
||||
# 更新时间(飙升/热度排序用;GitLink 字段名兜底多个)
|
||||
r["updated_at"] = (info.get("updated_at") or info.get("time")
|
||||
or info.get("updated_on") or r.get("updated_at"))
|
||||
if info.get("language") and isinstance(info["language"], dict):
|
||||
r["language"] = info["language"]
|
||||
contributors_map[fullname] = c.contributors(owner, name, limit=100)
|
||||
languages_map[fullname] = c.languages(owner, name)
|
||||
readmes[fullname] = c.readme(owner, name)[:4000]
|
||||
# 活跃讨论:开放的 Issue / PR(评论多的=热讨论)
|
||||
issues_map[fullname] = c.issues(owner, name, state="open", max_pages=1, page_size=50)
|
||||
prs_map[fullname] = c.prs(owner, name, state="open", max_pages=1, page_size=50)
|
||||
return {"repos": repos, "contributors_map": contributors_map,
|
||||
"languages_map": languages_map, "readmes": readmes,
|
||||
"issues_map": issues_map, "prs_map": prs_map}
|
||||
|
||||
|
||||
def _normalize_search_hit(r: dict, fullname: str) -> dict:
|
||||
"""把 search_repos 返回项归一化为 build_graph 期望的形状。"""
|
||||
lang_obj = r.get("language")
|
||||
if not isinstance(lang_obj, dict):
|
||||
lang_obj = {"name": c.as_str(lang_obj)}
|
||||
return {
|
||||
"fullname": fullname,
|
||||
"identifier": r.get("identifier", fullname.split("/")[-1]),
|
||||
"author": r.get("author") or {},
|
||||
"description": c.as_str(r.get("description")),
|
||||
"language": lang_obj,
|
||||
"praises_count": c.as_int(r.get("praises_count")),
|
||||
"forked_count": c.as_int(r.get("forked_count")),
|
||||
"forked_from_project_id": r.get("forked_from_project_id"),
|
||||
}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 渲染:Mermaid / DOT / Markdown 报告
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
_NODE_LIMIT = 40 # 防止 Mermaid 爆炸
|
||||
|
||||
_NODE_STYLE = {
|
||||
"repo": ("repoNode", "#4C78A8"),
|
||||
"scholar": ("scholarNode", "#F58518"),
|
||||
"topic": ("topicNode", "#54A24B"),
|
||||
}
|
||||
|
||||
|
||||
def _safe_id(nid: str) -> str:
|
||||
"""Mermaid/DOT 节点 id 用安全字符(去冒号斜杠)。"""
|
||||
return nid.replace(":", "_").replace("/", "_").replace("-", "_")
|
||||
|
||||
|
||||
def render_mermaid(result: dict[str, Any], node_limit: int = _NODE_LIMIT) -> str:
|
||||
"""渲染前 ~40 节点的 Mermaid graph TD(带 classDef 着色)。"""
|
||||
lines = ["```mermaid", "graph TD"]
|
||||
# 类定义
|
||||
for t, (cls, color) in _NODE_STYLE.items():
|
||||
lines.append(f" classDef {cls} fill:{color},stroke:#333,color:#fff;")
|
||||
|
||||
nodes = result.get("nodes", [])
|
||||
edges = result.get("edges", [])
|
||||
# 取前 node_limit 个节点(repo 优先,再 scholar,再 topic)
|
||||
type_order = {"repo": 0, "scholar": 1, "topic": 2}
|
||||
ordered = sorted(nodes, key=lambda n: (type_order.get(n["type"], 9), n["id"]))
|
||||
picked = ordered[:node_limit]
|
||||
picked_ids = {n["id"] for n in picked}
|
||||
|
||||
label_map: dict[str, str] = {}
|
||||
for n in picked:
|
||||
sid = _safe_id(n["id"])
|
||||
label = n["label"].replace('"', "'")
|
||||
lines.append(f' {sid}["{label}"]')
|
||||
label_map[n["id"]] = sid
|
||||
cls = _NODE_STYLE.get(n["type"], ("", ""))[0]
|
||||
if cls:
|
||||
lines.append(f" class {sid} {cls};")
|
||||
|
||||
# 只画两端都在 picked 内的边;去重(同源同目标同类只画一条)
|
||||
seen_edge: set[tuple] = set()
|
||||
for e in edges:
|
||||
if e["source"] not in picked_ids or e["target"] not in picked_ids:
|
||||
continue
|
||||
key = (e["source"], e["target"], e["type"])
|
||||
if key in seen_edge:
|
||||
continue
|
||||
seen_edge.add(key)
|
||||
a = label_map[e["source"]]
|
||||
b = label_map[e["target"]]
|
||||
w = e.get("weight", 1.0)
|
||||
et = e["type"]
|
||||
# 不同边类型用不同箭头标签
|
||||
lines.append(f' {a} -- "{et}({w:.2f})" --> {b}')
|
||||
|
||||
lines.append("```")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def render_dot(result: dict[str, Any], node_limit: int = _NODE_LIMIT) -> str:
|
||||
"""渲染 Graphviz DOT 字符串(带节点着色)。"""
|
||||
lines = ["digraph G {", ' rankdir=LR;',
|
||||
' graph [fontname="Helvetica"];',
|
||||
' node [fontname="Helvetica", style="filled"];',
|
||||
' edge [fontname="Helvetica"];']
|
||||
type_order = {"repo": 0, "scholar": 1, "topic": 2}
|
||||
nodes = result.get("nodes", [])
|
||||
edges = result.get("edges", [])
|
||||
ordered = sorted(nodes, key=lambda n: (type_order.get(n["type"], 9), n["id"]))
|
||||
picked = ordered[:node_limit]
|
||||
picked_ids = {n["id"] for n in picked}
|
||||
label_map: dict[str, str] = {}
|
||||
for n in picked:
|
||||
sid = _safe_id(n["id"])
|
||||
label = n["label"].replace('"', "'")
|
||||
color = _NODE_STYLE.get(n["type"], ("", "#CCCCCC"))[1]
|
||||
lines.append(f' {sid} [label="{label}", fillcolor="{color}"];')
|
||||
label_map[n["id"]] = sid
|
||||
for e in edges:
|
||||
if e["source"] not in picked_ids or e["target"] not in picked_ids:
|
||||
continue
|
||||
a = label_map[e["source"]]
|
||||
b = label_map[e["target"]]
|
||||
lines.append(f' {a} -> {b} [label="{e["type"]}"];')
|
||||
lines.append("}")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def render_report(result: dict[str, Any]) -> str:
|
||||
meta = result["meta"]
|
||||
heat = result.get("topic_heat", [])
|
||||
scholars = result.get("core_scholars", [])
|
||||
teams = result.get("core_teams", [])
|
||||
lines = [
|
||||
"# 科研热点追踪与知识图谱报告\n",
|
||||
f"> 场景 S2 · 子赛题四「应用 GitLink 辅助科研」\n",
|
||||
f"**关键词**: {', '.join(result.get('keywords') or []) or '—'}\n",
|
||||
"## 一、图谱概览\n",
|
||||
f"- 仓库节点: **{meta['repo_count']}**",
|
||||
f"- 学者节点: **{meta['scholar_count']}**",
|
||||
f"- 主题节点: **{meta['topic_count']}**",
|
||||
f"- 节点总数: **{meta['node_count']}**",
|
||||
f"- 边总数: **{meta['edge_count']}**\n",
|
||||
"## 二、主题热度榜(基于全部仓库 description)\n",
|
||||
"| 排名 | 主题 | 覆盖仓库数 |",
|
||||
"|------|------|-----------|",
|
||||
]
|
||||
if heat:
|
||||
for i, h in enumerate(heat, 1):
|
||||
lines.append(f"| {i} | `{h['topic']}` | {h['count']} |")
|
||||
else:
|
||||
lines.append("| — | (未识别到明确主题) | — |")
|
||||
lines += ["\n## 三、核心学者(按出现仓库数排序)\n",
|
||||
"| 排名 | 学者 | 关联仓库数 |",
|
||||
"|------|------|-----------|"]
|
||||
if scholars:
|
||||
for i, s in enumerate(scholars[:10], 1):
|
||||
lines.append(f"| {i} | `{s['login']}` | {s['repo_count']} |")
|
||||
else:
|
||||
lines.append("| — | (未识别到学者) | — |")
|
||||
lines += [f"\n## 四、核心团队(组织型仓库拥有者)\n"]
|
||||
if teams:
|
||||
lines += ["| 团队/组织 | 拥有仓库数 |", "|-----------|-----------|"]
|
||||
for t in teams:
|
||||
if isinstance(t, dict):
|
||||
lines.append(f"| `{t.get('login')}` | {t.get('repo_count', 0)} |")
|
||||
else:
|
||||
lines.append(f"| `{t}` | — |")
|
||||
else:
|
||||
lines.append("(该批仓库均由个人账号拥有,无组织型团队)")
|
||||
# 飙升/热门项目(热度分数排序;日均增星 velocity 作趋势代理)
|
||||
trending = result.get("trending_repos", [])
|
||||
lines += ["\n## 五、热门 / 飙升项目(热度排序)\n",
|
||||
"| 仓库 | 语言 | ★ | ⑂ | 日均★ | 最近更新 |",
|
||||
"|------|------|---:|---:|---:|----------|"]
|
||||
if trending:
|
||||
for t in trending[:10]:
|
||||
lines.append(f"| `{t['repo']}` | {t['language'] or '—'} | {t['stars']} | "
|
||||
f"{t['forks']} | {t['velocity']} | {t['updated'] or '—'} |")
|
||||
else:
|
||||
lines.append("| — | (未取到仓库) | | | | |")
|
||||
|
||||
# 活跃讨论
|
||||
active = result.get("active_discussions", [])
|
||||
lines += ["\n## 六、活跃讨论(评论最多的 Issue / PR)\n",
|
||||
"# | 类型 | 仓库 | 标题 | 评论 |", "|-|------|------|------|---:|"]
|
||||
if active:
|
||||
for i, a in enumerate(active[:10], 1):
|
||||
lines.append(f"| {i} | {a['type']} | `{a['repo']}` | {a['title'][:50]} | {a['comments']} |")
|
||||
else:
|
||||
lines.append("| — | | | (暂无明显热讨论) | |")
|
||||
|
||||
lines.append("\n_配套产物:graph.json(结构化)+ graph.mmd(Mermaid)+ "
|
||||
"graph.dot(Graphviz DOT)_\n")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description="S2 科研热点追踪与知识图谱")
|
||||
ap.add_argument("--keywords", required=True,
|
||||
help='逗号分隔的关键词,如 "deep learning,nlp"')
|
||||
ap.add_argument("--repos-limit", type=int, default=20,
|
||||
help="每关键词搜索后去重取 top N 仓库(默认 20)")
|
||||
ap.add_argument("--out", help="输出目录(写 graph.json/report.md/graph.mmd/graph.dot);"
|
||||
"省略则打印 JSON")
|
||||
args = ap.parse_args()
|
||||
|
||||
keywords = [k.strip() for k in args.keywords.split(",") if k.strip()]
|
||||
data = collect(keywords, repos_limit=args.repos_limit)
|
||||
result = build_graph(data["repos"], data["contributors_map"],
|
||||
data["languages_map"], data["readmes"], keywords=keywords)
|
||||
# 热点追踪两翼:飙升项目 + 活跃讨论(图谱之外的"追踪"信号)
|
||||
result["trending_repos"] = compute_trending(data["repos"])
|
||||
result["active_discussions"] = compute_active(data["issues_map"], data["prs_map"])
|
||||
|
||||
if args.out:
|
||||
os.makedirs(args.out, exist_ok=True)
|
||||
with open(os.path.join(args.out, "graph.json"), "w", encoding="utf-8") as f:
|
||||
json.dump(result, f, ensure_ascii=False, indent=2)
|
||||
with open(os.path.join(args.out, "report.md"), "w", encoding="utf-8") as f:
|
||||
f.write(render_report(result))
|
||||
with open(os.path.join(args.out, "graph.mmd"), "w", encoding="utf-8") as f:
|
||||
f.write(render_mermaid(result))
|
||||
with open(os.path.join(args.out, "graph.dot"), "w", encoding="utf-8") as f:
|
||||
f.write(render_dot(result))
|
||||
print(f"✓ S2 知识图谱完成 → {args.out}/graph.json | report.md | graph.mmd | graph.dot")
|
||||
print(f" 节点: {result['meta']['node_count']} 边: {result['meta']['edge_count']} "
|
||||
f"仓库: {result['meta']['repo_count']}")
|
||||
heat = result["topic_heat"][:5]
|
||||
print(f" Top 主题: {', '.join(h['topic'] for h in heat)}")
|
||||
else:
|
||||
print(json.dumps(result, ensure_ascii=False, indent=2))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
@ -0,0 +1,507 @@
|
|||
"""hotspot.py — 科研热点追踪(全栈重构版)。
|
||||
|
||||
输入一组科研关键词,从 GitLink 平台按关键词搜索相关仓库,对每个仓库:
|
||||
- 拉取 repo_info(stars / forks / 更新时间 / 贡献者)
|
||||
- 拉取 issue + pr 列表(活跃讨论按评论数排序)
|
||||
- 从 description + readme 抽取主题标签
|
||||
- 聚合学者/团队贡献网络
|
||||
|
||||
输出:hotspot.json(结构化数据)+ report.md(中文简报)。
|
||||
|
||||
取数与算法分离:collect() 在线取数,compute_*() 纯函数离线可测。
|
||||
|
||||
用法:
|
||||
python hotspot.py --keywords "deep learning,机器学习" --repos-limit 12 --out ./out
|
||||
python hotspot.py --keywords "computer vision" # 仅打印 JSON
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
from collections import Counter, defaultdict
|
||||
from typing import Any
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
import collect as c
|
||||
import topics as T
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 小工具
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _now_ts() -> float:
|
||||
return time.time()
|
||||
|
||||
|
||||
def _days_ago(ts: float) -> int:
|
||||
"""粗略计算距今多少天(非精确日历差,但足以排序)。"""
|
||||
return max(0, int((_now_ts() - ts) / 86400))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 热度评分(纯函数,离线可测)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def compute_trending_score(stars: int, forks: int, updated_days_ago: int) -> int:
|
||||
"""仓库热度综合评分。
|
||||
|
||||
公式:stars + forks × 2 + 近期更新加成(30 天内加成最高)。
|
||||
满分无上限,用于仓库间横向排序。
|
||||
"""
|
||||
base = stars + forks * 2
|
||||
if updated_days_ago <= 7:
|
||||
recency = 30
|
||||
elif updated_days_ago <= 30:
|
||||
recency = 20
|
||||
elif updated_days_ago <= 90:
|
||||
recency = 10
|
||||
elif updated_days_ago <= 180:
|
||||
recency = 5
|
||||
else:
|
||||
recency = 0
|
||||
return base + recency
|
||||
|
||||
|
||||
def compute_velocity(stars: int, updated_days_ago: int) -> float:
|
||||
"""日均星标增速(近似值)。"""
|
||||
if updated_days_ago <= 0:
|
||||
updated_days_ago = 1
|
||||
return round(stars / max(updated_days_ago, 1), 2)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 在线取数
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def collect(keywords: list[str], repos_limit: int = 12) -> dict:
|
||||
"""在线取数:多关键词搜索 → 去重 → 仓库详情 → issue/PR → 贡献者 → README。
|
||||
|
||||
返回原始数据字典,供 compute() 消费。
|
||||
"""
|
||||
# Step 1: 多关键词搜索
|
||||
seen: dict[str, dict] = {}
|
||||
for kw in keywords:
|
||||
results = c.search_repos(kw, limit=repos_limit)
|
||||
for proj in results:
|
||||
full = c.repo_fullname(proj)
|
||||
if not full:
|
||||
continue
|
||||
if full not in seen:
|
||||
seen[full] = {
|
||||
"fullname": full,
|
||||
"matched_keywords": [],
|
||||
"project": proj,
|
||||
"info": None,
|
||||
"issues": [],
|
||||
"prs": [],
|
||||
"contributors": [],
|
||||
"readme": "",
|
||||
"languages": {},
|
||||
}
|
||||
seen[full]["matched_keywords"].append(kw)
|
||||
|
||||
repos = list(seen.values())
|
||||
|
||||
# 先按 stars 粗排,截断到 repos_limit 再深度取数(节省 API 调用)
|
||||
def _star_sort_key(r: dict) -> int:
|
||||
p = r["project"]
|
||||
return -(c.as_int(p.get("praises_count") or p.get("stars_count") or 0))
|
||||
|
||||
repos.sort(key=_star_sort_key)
|
||||
repos = repos[:repos_limit]
|
||||
|
||||
# Step 2: 逐仓库取详情
|
||||
for r in repos:
|
||||
full = r["fullname"]
|
||||
parts = full.split("/", 1)
|
||||
if len(parts) != 2:
|
||||
continue
|
||||
owner, repo = parts[0], parts[1]
|
||||
|
||||
# repo_info
|
||||
info = c.repo_info(owner, repo)
|
||||
r["info"] = info if isinstance(info, dict) else {}
|
||||
|
||||
# issues(取 open 状态,限制 5 页以控制耗时)
|
||||
r["issues"] = c.issues(owner, repo, state="open", max_pages=3, page_size=30)
|
||||
|
||||
# PRs(取 open 状态)
|
||||
r["prs"] = c.prs(owner, repo, state="open", max_pages=3, page_size=30)
|
||||
|
||||
# contributors
|
||||
r["contributors"] = c.contributors(owner, repo)
|
||||
|
||||
# readme(前 4000 字符,仅用于主题抽取)
|
||||
r["readme"] = (c.readme(owner, repo) or "")[:4000]
|
||||
|
||||
# languages
|
||||
r["languages"] = c.languages(owner, repo)
|
||||
|
||||
return {"keywords": keywords, "repos": repos, "repos_limit": repos_limit}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 离线计算(纯函数,不联网)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def compute(raw: dict) -> dict:
|
||||
"""从 collect() 的原始数据计算出所有热点指标。
|
||||
|
||||
输入结构见 collect() 返回值;输出为标准 hotspot.json 结构。
|
||||
"""
|
||||
repos_raw = raw.get("repos") or []
|
||||
|
||||
# ------ trending_repos ------
|
||||
trending: list[dict] = []
|
||||
now = _now_ts()
|
||||
for r in repos_raw:
|
||||
info = r.get("info") or {}
|
||||
proj = r.get("project") or {}
|
||||
|
||||
stars = c.as_int(
|
||||
info.get("watchers_count")
|
||||
or info.get("praises_count")
|
||||
or proj.get("praises_count")
|
||||
or 0
|
||||
)
|
||||
forks = c.as_int(
|
||||
info.get("forked_count")
|
||||
or proj.get("forked_count")
|
||||
or 0
|
||||
)
|
||||
# 解析更新时间
|
||||
update_str = (
|
||||
info.get("full_last_update_time")
|
||||
or info.get("last_update_time")
|
||||
or proj.get("full_last_update_time")
|
||||
or proj.get("last_update_time")
|
||||
or ""
|
||||
)
|
||||
update_ts = _parse_ts(update_str)
|
||||
days = _days_ago(update_ts)
|
||||
score = compute_trending_score(stars, forks, days)
|
||||
velocity = compute_velocity(stars, days)
|
||||
language = ""
|
||||
lang_obj = info.get("language") or proj.get("language")
|
||||
if isinstance(lang_obj, dict):
|
||||
language = lang_obj.get("name") or ""
|
||||
elif isinstance(lang_obj, str):
|
||||
language = lang_obj
|
||||
|
||||
trending.append({
|
||||
"repo": r["fullname"],
|
||||
"description": (
|
||||
info.get("description")
|
||||
or proj.get("description")
|
||||
or ""
|
||||
),
|
||||
"language": language,
|
||||
"stars": stars,
|
||||
"forks": forks,
|
||||
"score": score,
|
||||
"velocity": velocity,
|
||||
"matched_keywords": r.get("matched_keywords", []),
|
||||
"updated": _fmt_ts(update_ts),
|
||||
"contributors_count": c.as_int(info.get("contributor_users_count") or 0),
|
||||
"releases_count": c.as_int(info.get("version_releases_count") or 0),
|
||||
})
|
||||
|
||||
trending.sort(key=lambda x: -x["score"])
|
||||
|
||||
# ------ active_discussions ------
|
||||
discussions: list[dict] = []
|
||||
for r in repos_raw:
|
||||
full = r["fullname"]
|
||||
# issues
|
||||
for iss in r.get("issues") or []:
|
||||
comments = c.as_int(iss.get("comment_count") or iss.get("comments") or 0)
|
||||
if comments > 0:
|
||||
discussions.append({
|
||||
"type": "issue",
|
||||
"repo": full,
|
||||
"title": iss.get("title") or "(无标题)",
|
||||
"number": iss.get("index") or iss.get("number") or "",
|
||||
"state": iss.get("state") or iss.get("status") or "open",
|
||||
"comments": comments,
|
||||
})
|
||||
# PRs
|
||||
for pr in r.get("prs") or []:
|
||||
comments = c.as_int(pr.get("comment_count") or pr.get("comments") or 0)
|
||||
if comments > 0:
|
||||
discussions.append({
|
||||
"type": "pr",
|
||||
"repo": full,
|
||||
"title": pr.get("title") or "(无标题)",
|
||||
"number": pr.get("index") or pr.get("number") or "",
|
||||
"state": pr.get("state") or pr.get("status") or "open",
|
||||
"comments": comments,
|
||||
})
|
||||
|
||||
discussions.sort(key=lambda x: -x["comments"])
|
||||
|
||||
# ------ topic_heat ------
|
||||
# 对所有仓库的 description + readme 跑 topic_counter
|
||||
texts = []
|
||||
for r in repos_raw:
|
||||
info = r.get("info") or {}
|
||||
proj = r.get("project") or {}
|
||||
desc = info.get("description") or proj.get("description") or ""
|
||||
texts.append(desc)
|
||||
readme = r.get("readme") or ""
|
||||
if readme:
|
||||
texts.append(readme)
|
||||
heat = T.topic_counter(texts)
|
||||
topic_heat = [
|
||||
{"topic": topic, "count": count}
|
||||
for topic, count in heat.most_common(10)
|
||||
]
|
||||
|
||||
# ------ core_scholars / core_teams ------
|
||||
scholar_repo_count: Counter = Counter()
|
||||
# 记录每个 scholar 关联的仓库名
|
||||
scholar_repos: dict[str, list[str]] = defaultdict(list)
|
||||
for r in repos_raw:
|
||||
full = r["fullname"]
|
||||
for contrib in r.get("contributors") or []:
|
||||
login = c.login_of(contrib)
|
||||
if not login or _is_bot(login):
|
||||
continue
|
||||
scholar_repo_count[login] += 1
|
||||
if full not in scholar_repos[login]:
|
||||
scholar_repos[login].append(full)
|
||||
|
||||
core_scholars = [
|
||||
{
|
||||
"login": login,
|
||||
"repo_count": count,
|
||||
"repos": scholar_repos.get(login, []),
|
||||
}
|
||||
for login, count in scholar_repo_count.most_common(12)
|
||||
]
|
||||
|
||||
# 核心团队:按 owner(仓库第一段)聚合
|
||||
org_repo: Counter = Counter()
|
||||
for r in repos_raw:
|
||||
full = r["fullname"]
|
||||
org = full.split("/")[0] if "/" in full else full
|
||||
org_repo[org] += 1
|
||||
|
||||
core_teams = [
|
||||
{"login": org, "repo_count": count}
|
||||
for org, count in org_repo.most_common(8)
|
||||
]
|
||||
|
||||
# ------ meta ------
|
||||
total_issues = sum(len(r.get("issues") or []) for r in repos_raw)
|
||||
total_prs = sum(len(r.get("prs") or []) for r in repos_raw)
|
||||
|
||||
return {
|
||||
"scenario": "hotspot",
|
||||
"keywords": raw.get("keywords") or [],
|
||||
"trending_repos": trending,
|
||||
"active_discussions": discussions,
|
||||
"topic_heat": topic_heat,
|
||||
"core_scholars": core_scholars,
|
||||
"core_teams": core_teams,
|
||||
"meta": {
|
||||
"repo_count": len(repos_raw),
|
||||
"issue_count": total_issues,
|
||||
"pr_count": total_prs,
|
||||
"scholar_count": len(scholar_repo_count),
|
||||
"discussion_count": len(discussions),
|
||||
"topic_count": len(topic_heat),
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 时间工具
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _parse_ts(v: Any) -> float:
|
||||
"""把 GitLink 时间戳/字符串尽量解析为 Unix 浮点秒。"""
|
||||
if v is None:
|
||||
return 0.0
|
||||
if isinstance(v, (int, float)):
|
||||
if v > 1_000_000_000_000:
|
||||
return v / 1000.0
|
||||
return float(v)
|
||||
s = str(v).strip()
|
||||
if not s:
|
||||
return 0.0
|
||||
# ISO 8601 格式
|
||||
for fmt in ("%Y-%m-%dT%H:%M:%S", "%Y-%m-%d %H:%M:%S",
|
||||
"%Y-%m-%dT%H:%M:%SZ", "%Y-%m-%d"):
|
||||
try:
|
||||
return time.mktime(time.strptime(s[:19] if len(s) >= 19 else s, fmt))
|
||||
except ValueError:
|
||||
continue
|
||||
return 0.0
|
||||
|
||||
|
||||
def _fmt_ts(ts: float) -> str:
|
||||
"""Unix 浮点秒 → 'YYYY-MM-DD' 字符串。"""
|
||||
if ts <= 0:
|
||||
return "—"
|
||||
try:
|
||||
return time.strftime("%Y-%m-%d", time.localtime(ts))
|
||||
except (ValueError, OSError):
|
||||
return "—"
|
||||
|
||||
|
||||
_BOT_HINTS = ("bot", "i-robot", "dependabot", "renovate", "semantic-release-bot")
|
||||
|
||||
|
||||
def _is_bot(login: str) -> bool:
|
||||
low = login.lower()
|
||||
return any(h in low for h in _BOT_HINTS)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 报告生成
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _report_md(output: dict) -> str:
|
||||
"""从热点 JSON 产出中文简报 Markdown。"""
|
||||
kw = ", ".join(output.get("keywords") or [])
|
||||
meta = output.get("meta") or {}
|
||||
lines = [
|
||||
f"# 🔬 科研热点追踪报告",
|
||||
f"",
|
||||
f"> 关键词:{kw}",
|
||||
f"> 扫描时间:{_fmt_ts(_now_ts())}",
|
||||
f"> 覆盖仓库:{meta.get('repo_count', 0)} 个 "
|
||||
f"· 讨论 {meta.get('discussion_count', 0)} 条 "
|
||||
f"· 主题 {meta.get('topic_count', 0)} 个 "
|
||||
f"· 学者 {meta.get('scholar_count', 0)} 位",
|
||||
f"",
|
||||
]
|
||||
|
||||
# 飙升项目 top 5
|
||||
trending = output.get("trending_repos") or []
|
||||
if trending:
|
||||
lines.append("## 🔥 飙升项目 Top 5")
|
||||
lines.append("")
|
||||
lines.append("| # | 仓库 | 语言 | ★ Star | ⑂ Fork | 热度 | 更新 |")
|
||||
lines.append("|---|------|------|--------|--------|------|------|")
|
||||
for i, r in enumerate(trending[:5], 1):
|
||||
lines.append(
|
||||
f"| {i} | `{r['repo']}` | {r['language'] or '—'} | "
|
||||
f"{r['stars']} | {r['forks']} | {r['score']} | {r['updated']} |"
|
||||
)
|
||||
lines.append("")
|
||||
|
||||
# 活跃讨论 top 5
|
||||
discussions = output.get("active_discussions") or []
|
||||
if discussions:
|
||||
lines.append("## 💬 活跃讨论 Top 5")
|
||||
lines.append("")
|
||||
for i, d in enumerate(discussions[:5], 1):
|
||||
tp = "🐛 Issue" if d["type"] == "issue" else "🔀 PR"
|
||||
lines.append(
|
||||
f"{i}. {tp} [{d['repo']}] {d['title']} "
|
||||
f"(#{d['number']} · {d['comments']} 💬)"
|
||||
)
|
||||
lines.append("")
|
||||
|
||||
# 热门主题
|
||||
topic_heat = output.get("topic_heat") or []
|
||||
if topic_heat:
|
||||
lines.append("## 📊 热门主题")
|
||||
lines.append("")
|
||||
for t in topic_heat:
|
||||
bar = "█" * min(t["count"], 20)
|
||||
lines.append(f"- **{t['topic']}** — {t['count']} 个仓库 {bar}")
|
||||
lines.append("")
|
||||
|
||||
# 核心学者
|
||||
scholars = output.get("core_scholars") or []
|
||||
if scholars:
|
||||
lines.append("## 👥 核心学者")
|
||||
lines.append("")
|
||||
for s in scholars[:5]:
|
||||
lines.append(f"- **{s['login']}** — 关联 {s['repo_count']} 个仓库")
|
||||
lines.append("")
|
||||
|
||||
# 核心团队
|
||||
teams = output.get("core_teams") or []
|
||||
if teams:
|
||||
lines.append("## 🏛 活跃组织/团队")
|
||||
lines.append("")
|
||||
for t in teams:
|
||||
lines.append(f"- **{t['login']}** — {t['repo_count']} 个仓库")
|
||||
lines.append("")
|
||||
|
||||
lines.append("---")
|
||||
lines.append("*由 gitlink-research-hotspot 自动生成*")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# CLI 入口
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def main() -> None:
|
||||
ap = argparse.ArgumentParser(
|
||||
description="科研热点追踪 — 多角度扫描 GitLink 科研仓库"
|
||||
)
|
||||
ap.add_argument(
|
||||
"--keywords", "-k",
|
||||
default="deep learning,机器学习",
|
||||
help="搜索关键词,逗号分隔(默认 deep learning,机器学习)",
|
||||
)
|
||||
ap.add_argument(
|
||||
"--repos-limit", type=int, default=12,
|
||||
help="最大分析仓库数(默认 12)",
|
||||
)
|
||||
ap.add_argument(
|
||||
"--out", "-o",
|
||||
default="",
|
||||
help="输出目录(不传则仅打印 JSON 到 stdout)",
|
||||
)
|
||||
args = ap.parse_args()
|
||||
|
||||
kw_list = [k.strip() for k in args.keywords.split(",") if k.strip()]
|
||||
if not kw_list:
|
||||
print(json.dumps({"ok": False, "error": "keywords required"}, ensure_ascii=False))
|
||||
sys.exit(1)
|
||||
|
||||
# 取数
|
||||
sys.stderr.write(f"[hotspot] 关键词: {kw_list} 上限: {args.repos_limit}\n")
|
||||
sys.stderr.flush()
|
||||
raw = collect(kw_list, args.repos_limit)
|
||||
|
||||
# 计算
|
||||
sys.stderr.write(f"[hotspot] 仓库: {len(raw['repos'])} 计算热点…\n")
|
||||
sys.stderr.flush()
|
||||
output = compute(raw)
|
||||
|
||||
json_text = json.dumps(output, ensure_ascii=False, indent=2)
|
||||
|
||||
if args.out:
|
||||
os.makedirs(args.out, exist_ok=True)
|
||||
json_path = os.path.join(args.out, "hotspot.json")
|
||||
with open(json_path, "w", encoding="utf-8") as f:
|
||||
f.write(json_text)
|
||||
report_path = os.path.join(args.out, "report.md")
|
||||
with open(report_path, "w", encoding="utf-8") as f:
|
||||
f.write(_report_md(output))
|
||||
sys.stderr.write(
|
||||
f"[hotspot] ✓ 完成 "
|
||||
f"仓库={output['meta']['repo_count']} "
|
||||
f"讨论={output['meta']['discussion_count']} "
|
||||
f"主题={output['meta']['topic_count']} "
|
||||
f"学者={output['meta']['scholar_count']}\n"
|
||||
)
|
||||
sys.stderr.write(f"[hotspot] 产物: {json_path}, {report_path}\n")
|
||||
sys.stderr.flush()
|
||||
else:
|
||||
print(json_text)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
@ -0,0 +1,474 @@
|
|||
"""lineage.py — S1 仓库级科研项目洞悉。
|
||||
|
||||
对一个 GitLink 科研仓库做「项目谱系(lineage)」分析:从默认分支提交时间线、
|
||||
合并 PR 的演进模式、文档演进、实验/评测文件组织、关键创新点五个角度,回答
|
||||
「这个科研项目是怎么一步步长成现在这样的、值得引用/复现到哪一步」。
|
||||
|
||||
技术栈 = Go 出数据(经 gitlink-cli + collect.py 取数)+ Python 做算法(本文件)。
|
||||
所有 GitLink 操作经 gitlink-cli,禁止用 gh/glab(见 gitlink-shared)。
|
||||
|
||||
用法:
|
||||
python lineage.py --owner mindspore-Ecosystem --repo mindspore --out ./out
|
||||
python lineage.py --owner O --repo R # 仅打印 JSON
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from collections import Counter
|
||||
from typing import Any
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
import collect as c # noqa: E402
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 提交时间解析(commit timestamp 可能是 ISO 字符串或整数秒)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _to_epoch(v: Any) -> float:
|
||||
"""把 GitLink commit 的 timestamp(ISO 字符串或整数秒)统一解析为 epoch 秒。
|
||||
|
||||
失败返回 0.0。
|
||||
"""
|
||||
if v is None:
|
||||
return 0.0
|
||||
if isinstance(v, (int, float)):
|
||||
# 毫秒级时间戳兜底(GitLink 多为秒)
|
||||
return float(v) / 1000.0 if v > 1e12 else float(v)
|
||||
s = str(v).strip()
|
||||
if not s:
|
||||
return 0.0
|
||||
# 纯数字串
|
||||
if s.isdigit():
|
||||
f = float(s)
|
||||
return f / 1000.0 if f > 1e12 else f
|
||||
# ISO 8601:'2024-05-01T08:00:00Z' / '2024-05-01 08:00:00'
|
||||
s = s.replace("Z", "+00:00")
|
||||
try:
|
||||
import datetime as _dt
|
||||
return _dt.datetime.fromisoformat(s).timestamp()
|
||||
except (ValueError, TypeError):
|
||||
# 退而求其次:抽首个 YYYY-MM-DD
|
||||
m = re.search(r"(\d{4})-(\d{2})-(\d{2})", str(v))
|
||||
if m:
|
||||
try:
|
||||
import datetime as _dt
|
||||
return _dt.datetime(int(m.group(1)), int(m.group(2)),
|
||||
int(m.group(3))).timestamp()
|
||||
except ValueError:
|
||||
return 0.0
|
||||
return 0.0
|
||||
|
||||
|
||||
def _iso_date(epoch: float) -> str:
|
||||
"""epoch 秒 → 'YYYY-MM-DD' 字符串(0 → '')。"""
|
||||
if not epoch:
|
||||
return ""
|
||||
import datetime as _dt
|
||||
try:
|
||||
return _dt.datetime.utcfromtimestamp(epoch).strftime("%Y-%m-%d")
|
||||
except (OSError, ValueError, OverflowError):
|
||||
return ""
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 分类器:实验/评测文件 vs 文档文件
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# 命中即判为「实验/评测/数据」类文件(科研产物信号)。
|
||||
# 匹配「目录段或文件名」:以 experiment/benchmark/eval/test(s)/data 开头
|
||||
# (后接 [\w-]* 续写,或紧接分隔符/串尾)。
|
||||
_EXPERIMENT_RE = re.compile(
|
||||
r"(^|[_/\-])(experiment[\w-]*|benchmark[\w-]*|eval[\w-]*|tests?|data)([/\-\._]|$)",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
def is_experiment_file(path: str) -> bool:
|
||||
"""判断文件路径是否属于「实验/评测/数据」类(科研产物)。
|
||||
|
||||
命中 experiment*/benchmark*/eval*/tests?/data/ 任意一段(作为目录名或文件名前缀)
|
||||
返回 True。例如:
|
||||
- experiments/run.py, experiment_train.py, benchmark/eval.py
|
||||
- tests/test_x.py, data/dataset.csv, src/benchmark_infer.py
|
||||
"""
|
||||
if not path:
|
||||
return False
|
||||
return bool(_EXPERIMENT_RE.search(path))
|
||||
|
||||
|
||||
def is_doc_file(path: str) -> bool:
|
||||
"""判断文件路径是否属于文档类:*.md(任意位置)或 docs/ 下任意文件。"""
|
||||
if not path:
|
||||
return False
|
||||
low = path.lower()
|
||||
if low.endswith(".md"):
|
||||
return True
|
||||
return low.startswith("docs/") or ("/docs/" in low)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 分支地图(单分支简化:默认分支为唯一分支)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def build_branch_map(commits: list[dict], default_branch: str) -> list[dict]:
|
||||
"""构建分支活跃度地图。本场景单分支简化:默认分支为唯一分支。
|
||||
|
||||
返回 [{name, commits, last_active, is_default}]。
|
||||
"""
|
||||
n = len(commits) if isinstance(commits, list) else 0
|
||||
last = 0.0
|
||||
for cm in commits or []:
|
||||
t = _to_epoch(cm.get("timestamp"))
|
||||
if t and t > last:
|
||||
last = t
|
||||
return [{
|
||||
"name": default_branch or "master",
|
||||
"commits": n,
|
||||
"last_active": _iso_date(last),
|
||||
"is_default": True,
|
||||
}]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 提交时间线(按周/按日聚合)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def commit_timeline(commits: list[dict], bucket: str = "day") -> list[dict]:
|
||||
"""把提交按日期聚合为时间线,返回按日期升序的 [{date, count}]。
|
||||
|
||||
bucket ∈ {'day','week'};week 以 ISO 年-周 表示。
|
||||
"""
|
||||
counter: Counter = Counter()
|
||||
for cm in commits or []:
|
||||
t = _to_epoch(cm.get("timestamp"))
|
||||
if not t:
|
||||
continue
|
||||
if bucket == "week":
|
||||
import datetime as _dt
|
||||
iso = _dt.datetime.utcfromtimestamp(t).isocalendar()
|
||||
key = f"{iso[0]}-W{iso[1]:02d}"
|
||||
else:
|
||||
key = _iso_date(t)
|
||||
counter[key] += 1
|
||||
return [{"date": k, "count": counter[k]} for k in sorted(counter)]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 合并 PR 的演进模式
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def pr_merge_patterns(merged_prs: list[dict]) -> list[dict]:
|
||||
"""从已合并 PR 抽取演进模式。
|
||||
|
||||
返回 [{number, title, status, merged_time, changed_files}]。
|
||||
changed_files 取 pr 提供的 changed_files / changedFiles / additions/deletions 的近似。
|
||||
"""
|
||||
out: list[dict] = []
|
||||
for pr in merged_prs or []:
|
||||
if not isinstance(pr, dict):
|
||||
continue
|
||||
# 时间:优先 merged 时间字段,否则 pr_created_unix
|
||||
merged_t = (pr.get("pr_merged_unix") or pr.get("merged_at")
|
||||
or pr.get("pr_updated_unix") or pr.get("pr_created_unix"))
|
||||
# 文件改动数:优先详情接口的 files_count(列表 API 不返回)
|
||||
changed = (pr.get("files_count") or pr.get("changed_files")
|
||||
or pr.get("changedFiles") or pr.get("file_nums") or 0)
|
||||
# 兜底:用 additions/deletions 之和近似
|
||||
if not changed:
|
||||
changed = c.as_int(pr.get("additions"), 0) + c.as_int(pr.get("deletions"), 0)
|
||||
# 标题只取首行 + 截断到 80 字(有些 PR 创建时把正文塞进了 title 字段)
|
||||
raw_title = c.as_str(pr.get("title")).split("\n", 1)[0].strip()
|
||||
title = raw_title[:80]
|
||||
out.append({
|
||||
"number": pr.get("index") or pr.get("number") or pr.get("id"),
|
||||
"title": title,
|
||||
"status": pr.get("status"),
|
||||
"merged_time": _iso_date(_to_epoch(merged_t)),
|
||||
"changed_files": c.as_int(changed),
|
||||
})
|
||||
# 按合并时间升序(空时间排末尾)
|
||||
out.sort(key=lambda x: (x["merged_time"] == "", x["merged_time"]))
|
||||
return out
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 文档演进(docs/*.md 的近似最后修改信息)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def doc_evolution(tree_entries: list[dict]) -> list[dict]:
|
||||
"""从仓库树抽取 docs/ 下的文档清单,近似其最后修改日期。
|
||||
|
||||
tree 来自 c.tree(owner, repo, path='docs')。每条 entry 形如
|
||||
{name, path, type, ...}。最后修改日期在树端点通常不可得,用文件名中的
|
||||
日期或留空(report 中标注「近似」)。
|
||||
返回 [{file, last_date}]。
|
||||
"""
|
||||
out: list[dict] = []
|
||||
for e in tree_entries or []:
|
||||
if not isinstance(e, dict):
|
||||
continue
|
||||
name = c.as_str(e.get("name") or e.get("path"))
|
||||
path = c.as_str(e.get("path") or name)
|
||||
if not is_doc_file(path):
|
||||
continue
|
||||
last = c.as_str(e.get("last_commit") or e.get("commit_date")
|
||||
or e.get("date"))
|
||||
if not last:
|
||||
m = re.search(r"(\d{4})-(\d{2})-(\d{2})", path)
|
||||
last = m.group(0) if m else ""
|
||||
out.append({"file": name, "last_date": last})
|
||||
out.sort(key=lambda x: (x["last_date"] == "", x["file"]))
|
||||
return out
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 创新点识别(高影响合并)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def innovation_points(merged_prs: list[dict], commits: list[dict],
|
||||
top: int = 8) -> list[dict]:
|
||||
"""识别高影响合并作为项目的创新/里程碑点。
|
||||
|
||||
判据(任一):
|
||||
- 改动文件数高(>= 中位数的 1.5 倍,或绝对值 >= 10)→ 「大规模重构/新特性」
|
||||
- 标题含里程碑关键词(add/implement/feature/release/benchmark/...)→ 「特性引入」
|
||||
返回 [{description, evidence, category}],按影响度降序取前 top。
|
||||
"""
|
||||
patterns = pr_merge_patterns(merged_prs)
|
||||
if not patterns:
|
||||
return []
|
||||
|
||||
files = [p["changed_files"] for p in patterns if p["changed_files"] > 0]
|
||||
median = sorted(files)[len(files) // 2] if files else 0
|
||||
|
||||
MILESTONE_RE = re.compile(
|
||||
r"(add|implement|support|feature|release|benchmark|refactor|"
|
||||
r"experiment|dataset|train|inference|v\d+\.\d+)", re.IGNORECASE)
|
||||
|
||||
scored: list[tuple[float, dict]] = []
|
||||
for p in patterns:
|
||||
cf = p["changed_files"]
|
||||
title = p["title"]
|
||||
category = ""
|
||||
impact = float(cf)
|
||||
if cf >= 10 or (median and cf >= median * 1.5):
|
||||
category = "大规模重构/新特性"
|
||||
impact += 10
|
||||
if MILESTONE_RE.search(title):
|
||||
category = category or "特性引入"
|
||||
impact += 5
|
||||
if not category:
|
||||
continue
|
||||
evidence = (f"PR #{p['number']} 「{title[:48]}」 "
|
||||
f"改动 {cf} 文件,合并于 {p['merged_time'] or '未知时间'}")
|
||||
scored.append((impact, {
|
||||
"description": title.strip() or f"PR #{p['number']}",
|
||||
"evidence": evidence,
|
||||
"category": category,
|
||||
}))
|
||||
scored.sort(key=lambda x: -x[0])
|
||||
return [item for _, item in scored[:top]]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 主流程:取数 + 算法
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def lineage(owner: str, repo: str, branches_limit: int = 5) -> dict[str, Any]:
|
||||
"""取数 + 分析,返回完整 lineage 结果 dict。"""
|
||||
info = c.repo_info(owner, repo)
|
||||
default_branch = info.get("default_branch") or "master"
|
||||
|
||||
commits = c.commits(owner, repo, ref=default_branch, max_pages=10, page_size=100)
|
||||
# 已合并 PR:state=merged(collect 透传)
|
||||
merged_prs = c.prs(owner, repo, state="merged", max_pages=10, page_size=50)
|
||||
# 列表 API 不返回文件改动数 → 逐个取详情补 files_count(限速 + 上限 30 个,控 API 调用)
|
||||
for pr in merged_prs[:30]:
|
||||
idx = pr.get("index") or pr.get("number") or pr.get("id")
|
||||
if idx is None:
|
||||
continue
|
||||
try:
|
||||
det = c.pr_detail(owner, repo, int(idx))
|
||||
except (TypeError, ValueError):
|
||||
det = {}
|
||||
if det:
|
||||
if det.get("files_count") is not None:
|
||||
pr["files_count"] = det.get("files_count")
|
||||
if det.get("commits_count") is not None:
|
||||
pr["commits_count"] = det.get("commits_count")
|
||||
tree_root = c.tree(owner, repo, ref=default_branch)
|
||||
docs_tree = c.tree(owner, repo, path="docs", ref=default_branch)
|
||||
_readme = c.readme(owner, repo, ref=default_branch)
|
||||
|
||||
# 扫描整棵树,挑出实验/评测文件
|
||||
exp_files: list[str] = []
|
||||
all_tree = (tree_root or []) + (docs_tree or [])
|
||||
for e in all_tree:
|
||||
if not isinstance(e, dict):
|
||||
continue
|
||||
p = c.as_str(e.get("path") or e.get("name"))
|
||||
if p and is_experiment_file(p):
|
||||
exp_files.append(p)
|
||||
|
||||
timeline = commit_timeline(commits, bucket="day")
|
||||
branch_map = build_branch_map(commits, default_branch)[:branches_limit]
|
||||
pr_patterns = pr_merge_patterns(merged_prs)
|
||||
docs = doc_evolution(docs_tree if docs_tree else tree_root)
|
||||
innovations = innovation_points(merged_prs, commits)
|
||||
|
||||
return {
|
||||
"scenario": "S1_repository_research_insight",
|
||||
"repo": f"{owner}/{repo}",
|
||||
"default_branch": default_branch,
|
||||
"commit_timeline": timeline,
|
||||
"branch_map": branch_map,
|
||||
"pr_merge_patterns": pr_patterns,
|
||||
"doc_evolution": docs,
|
||||
"experiment_files": sorted(set(exp_files)),
|
||||
"innovation_points": innovations,
|
||||
"meta": {
|
||||
"commit_count": len(commits),
|
||||
"merged_pr_count": len(merged_prs),
|
||||
"doc_count": len(docs),
|
||||
"experiment_file_count": len(exp_files),
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 渲染:Markdown 报告 + Mermaid gitGraph
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def render_report(result: dict[str, Any]) -> str:
|
||||
repo = result["repo"]
|
||||
meta = result["meta"]
|
||||
lines = [
|
||||
f"# 仓库级科研项目洞悉报告 — {repo}\n",
|
||||
f"> 场景 S1 · 子赛题四「应用 GitLink 辅助科研」· 项目谱系(lineage)分析\n",
|
||||
"## 一、基础信息\n",
|
||||
f"- **默认分支**: `{result['default_branch']}`",
|
||||
f"- **采样提交**: {meta['commit_count']} 条(默认分支,最多 10×100)",
|
||||
f"- **已合并 PR**: {meta['merged_pr_count']} 个",
|
||||
f"- **文档文件**: {meta['doc_count']} 个",
|
||||
f"- **实验/评测文件**: {meta['experiment_file_count']} 个\n",
|
||||
]
|
||||
|
||||
lines.append("## 二、提交活跃度时间线\n")
|
||||
tl = result["commit_timeline"]
|
||||
if tl:
|
||||
peak = max(tl, key=lambda x: x["count"])
|
||||
lines.append(f"- 时间跨度: {tl[0]['date']} → {tl[-1]['date']}"
|
||||
f"(共 {len(tl)} 个有提交的日期)")
|
||||
lines.append(f"- 峰值: {peak['date']} 当日 {peak['count']} 次提交\n")
|
||||
else:
|
||||
lines.append("- (未取到提交时间线)\n")
|
||||
|
||||
lines.append("## 三、分支地图\n")
|
||||
lines.append("| 分支 | 提交数 | 最后活跃 | 是否默认 |")
|
||||
lines.append("|------|:------:|----------|:--------:|")
|
||||
for b in result["branch_map"]:
|
||||
lines.append(f"| `{b['name']}` | {b['commits']} | {b['last_active'] or '—'} |"
|
||||
f" {'是' if b['is_default'] else '否'} |")
|
||||
lines.append("")
|
||||
|
||||
lines.append("## 四、合并 PR 演进模式(高影响合并预览)\n")
|
||||
prs = result["pr_merge_patterns"]
|
||||
if prs:
|
||||
lines.append("| PR | 标题 | 改动文件 | 合并时间 |")
|
||||
lines.append("|----|------|:--------:|----------|")
|
||||
for p in prs[:10]:
|
||||
lines.append(f"| #{p['number']} | {p['title']} | "
|
||||
f"{p['changed_files']} | {p['merged_time'] or '—'} |")
|
||||
else:
|
||||
lines.append("- (无已合并 PR)")
|
||||
lines.append("")
|
||||
|
||||
lines.append("## 五、创新/里程碑点\n")
|
||||
inno = result["innovation_points"]
|
||||
if inno:
|
||||
for i, it in enumerate(inno, 1):
|
||||
lines.append(f"{i}. **[{it['category']}]** {it['description']}")
|
||||
lines.append(f" - 证据: {it['evidence']}")
|
||||
else:
|
||||
lines.append("- (未识别到明显高影响合并)")
|
||||
lines.append("")
|
||||
|
||||
lines.append("## 六、文档演进(docs/*)\n")
|
||||
docs = result["doc_evolution"]
|
||||
if docs:
|
||||
lines.append("| 文档 | 近似最后日期 |")
|
||||
lines.append("|------|--------------|")
|
||||
for d in docs[:15]:
|
||||
lines.append(f"| {d['file']} | {d['last_date'] or '—'} |")
|
||||
else:
|
||||
lines.append("- (docs/ 下无文档或树不可得)")
|
||||
lines.append("")
|
||||
|
||||
lines.append("## 七、实验/评测文件组织\n")
|
||||
exps = result["experiment_files"]
|
||||
if exps:
|
||||
for p in exps[:20]:
|
||||
lines.append(f"- `{p}`")
|
||||
if len(exps) > 20:
|
||||
lines.append(f"- ...(共 {len(exps)} 个,此处仅列前 20)")
|
||||
else:
|
||||
lines.append("- (未在仓库树中识别到 experiment/benchmark/eval/test/data 目录)")
|
||||
lines.append("")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def render_mermaid(result: dict[str, Any]) -> str:
|
||||
"""渲染 Mermaid gitGraph:以默认分支为唯一分支,按合并时间标里程碑 PR。"""
|
||||
lines = ["```mermaid", "gitGraph"]
|
||||
lines.append(f" commit id: \"{result['default_branch']} 起点\"")
|
||||
prs = result["pr_merge_patterns"]
|
||||
inno = result["innovation_points"]
|
||||
inno_titles = {it["description"] for it in inno}
|
||||
# 在时间线上穿插 commit / merge 标记(每 ~3 个 PR 一个 merge,里程碑标 hotfix)
|
||||
step = 0
|
||||
for p in prs:
|
||||
lines.append(" commit")
|
||||
step += 1
|
||||
if step % 3 == 0:
|
||||
tag = " 创新点" if any(p["title"].strip() == t for t in inno_titles) else ""
|
||||
lines.append(f" commit id: \"#{p['number']}{tag}\" tag: \"{p['merged_time'] or 'PR'}\"")
|
||||
lines.append("```")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description="S1 仓库级科研项目洞悉(lineage 谱系分析)")
|
||||
ap.add_argument("--owner", required=True)
|
||||
ap.add_argument("--repo", required=True)
|
||||
ap.add_argument("--branches-limit", type=int, default=5,
|
||||
help="分支地图上限(本场景单分支简化,默认 5)")
|
||||
ap.add_argument("--out", help="输出目录(写 lineage.json/report.md/branch_graph.mmd);"
|
||||
"省略则打印 JSON")
|
||||
args = ap.parse_args()
|
||||
|
||||
result = lineage(args.owner, args.repo, branches_limit=args.branches_limit)
|
||||
|
||||
if args.out:
|
||||
os.makedirs(args.out, exist_ok=True)
|
||||
with open(os.path.join(args.out, "lineage.json"), "w", encoding="utf-8") as f:
|
||||
json.dump(result, f, ensure_ascii=False, indent=2)
|
||||
with open(os.path.join(args.out, "report.md"), "w", encoding="utf-8") as f:
|
||||
f.write(render_report(result))
|
||||
with open(os.path.join(args.out, "branch_graph.mmd"), "w", encoding="utf-8") as f:
|
||||
f.write(render_mermaid(result))
|
||||
print(f"✓ S1 项目洞悉完成 → {args.out}/lineage.json | report.md | branch_graph.mmd")
|
||||
print(f" 提交 {result['meta']['commit_count']} 条 | "
|
||||
f"合并 PR {result['meta']['merged_pr_count']} 个 | "
|
||||
f"创新点 {len(result['innovation_points'])} 个")
|
||||
else:
|
||||
print(json.dumps(result, ensure_ascii=False, indent=2))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
@ -0,0 +1,320 @@
|
|||
"""match.py — S4 科研协作智能匹配。
|
||||
|
||||
输入一个科研仓库,分析其技术缺口(未解决 Issue 的主题/语言、开放 PR、研究空缺),
|
||||
再从 GitLink 平台候选池(本仓库贡献者 + 按缺口主题搜索到的用户)中,用「主题向量 + 语言匹配 +
|
||||
活跃度 + 协作开放度」综合打分,推荐最合适的跨团队/跨学者协作伙伴。
|
||||
|
||||
数据全部经 gitlink-cli 获取(issue +list / repo +contributors / repo +list --user / search +users)。
|
||||
|
||||
用法:
|
||||
python match.py --owner mindspore-Ecosystem --repo mindspore --top 10 --out ./out
|
||||
python match.py --owner O --repo R --format json # 仅打印 JSON
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
import sys
|
||||
from collections import Counter
|
||||
from typing import Any
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
import collect as c # noqa: E402
|
||||
import topics as T # noqa: E402
|
||||
|
||||
# 优先级 → 权重(Issue 缺口信号加权)
|
||||
PRIORITY_WEIGHT = {"urgent": 3, "high": 3, "紧急": 3, "高": 3,
|
||||
"normal": 2, "medium": 2, "普通": 2, "中": 2,
|
||||
"low": 1, "低": 1}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 向量与打分
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def cosine(c1: dict[str, float], c2: dict[str, float]) -> float:
|
||||
keys = set(c1) | set(c2)
|
||||
dot = sum(c1.get(k, 0.0) * c2.get(k, 0.0) for k in keys)
|
||||
n1 = math.sqrt(sum(v * v for v in c1.values()))
|
||||
n2 = math.sqrt(sum(v * v for v in c2.values()))
|
||||
return dot / (n1 * n2) if n1 and n2 else 0.0
|
||||
|
||||
|
||||
def jaccard(a: list[str], b: list[str]) -> float:
|
||||
sa, sb = set(a), set(b)
|
||||
if not sa or not sb:
|
||||
return 0.0
|
||||
return len(sa & sb) / len(sa | sb)
|
||||
|
||||
|
||||
def _priority_weight(issue: dict) -> float:
|
||||
p = (issue.get("priority_name") or issue.get("priority") or "").lower()
|
||||
for k, w in PRIORITY_WEIGHT.items():
|
||||
if k in str(p):
|
||||
return float(w)
|
||||
return 1.0
|
||||
|
||||
|
||||
def _parse_ratio(v: Any) -> float:
|
||||
"""把 '1.18%' / '0.2' / 0.2 等统一解析为 0~1 比例。"""
|
||||
if v is None:
|
||||
return 0.0
|
||||
s = str(v).strip()
|
||||
pct = s.endswith("%")
|
||||
if pct:
|
||||
s = s[:-1]
|
||||
try:
|
||||
f = float(s)
|
||||
except ValueError:
|
||||
return 0.0
|
||||
return f / 100.0 if (pct or f > 1.0) else f
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 缺口信号
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def build_gap_signals(owner: str, repo: str, info: dict,
|
||||
issue_sample: int = 100) -> tuple[Counter, list[str], list[dict]]:
|
||||
"""返回 (缺口主题词频 Counter, 需求语言列表, 缺口信号明细)。"""
|
||||
gap_topics: Counter = Counter()
|
||||
gap_langs: set[str] = set()
|
||||
signals: list[dict] = []
|
||||
|
||||
# 1) 仓库自身主题/语言(协作者应具备的基础方向)
|
||||
desc = (info.get("description") or "") + " " + c.readme(owner, repo)[:4000]
|
||||
for tp in T.extract_topics(desc):
|
||||
gap_topics[tp] += 1
|
||||
for lg in c.languages(owner, repo):
|
||||
gap_langs.add(lg)
|
||||
for lg in T.extract_languages(desc):
|
||||
gap_langs.add(lg)
|
||||
|
||||
# 2) 未解决 Issue 的主题/语言(核心缺口)
|
||||
open_issues = c.issues(owner, repo, state="open", max_pages=max(1, issue_sample // 50),
|
||||
page_size=50)
|
||||
for iss in open_issues[:issue_sample]:
|
||||
text = (iss.get("subject") or iss.get("title") or "")
|
||||
w = _priority_weight(iss)
|
||||
for tp in T.extract_topics(text):
|
||||
gap_topics[tp] += w
|
||||
for lg in T.extract_languages(text):
|
||||
gap_langs.add(lg)
|
||||
# 取每条 issue 的首个主题作为该条的证据
|
||||
tps = T.extract_topics(text)
|
||||
if tps:
|
||||
signals.append({"type": "unresolved_issue", "topic": tps[0],
|
||||
"evidence": text[:80], "priority": iss.get("priority_name", "")})
|
||||
|
||||
# 3) 开放 PR 正在推进的方向(轻量加成)
|
||||
for pr in c.prs(owner, repo, state="open", max_pages=1, page_size=30):
|
||||
text = (pr.get("title") or "") + " " + (pr.get("body") or "")
|
||||
for tp in T.extract_topics(text):
|
||||
gap_topics[tp] += 0.5
|
||||
|
||||
needed_langs = sorted(gap_langs)
|
||||
return gap_topics, needed_langs, signals
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 候选人画像与匹配
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def candidate_pool(owner: str, repo: str, gap_topics: Counter, pool_cap: int) -> list[str]:
|
||||
"""候选人 login 池:本仓库贡献者 + 按缺口主题搜到的外部用户。"""
|
||||
seen: list[str] = []
|
||||
seen_set: set[str] = set()
|
||||
|
||||
for contrib in c.contributors(owner, repo):
|
||||
login = c.login_of(contrib)
|
||||
# 过滤明显机器人账号
|
||||
if login and login not in seen_set and "bot" not in login.lower() and login.lower() != "i-robot":
|
||||
seen.append(login)
|
||||
seen_set.add(login)
|
||||
if len(seen) >= pool_cap:
|
||||
return seen
|
||||
|
||||
# 取词频最高的若干主题,用其英文关键词搜外部用户
|
||||
top_topics = [t for t, _ in gap_topics.most_common(5)]
|
||||
eng_kw = {"nlp": "nlp", "deep_learning": "deep learning", "computer_vision": "cv",
|
||||
"reinforcement_learning": "reinforcement learning",
|
||||
"graph_learning": "gnn", "federated_learning": "federated",
|
||||
"scientific_computing": "cuda", "data_mining": "machine learning",
|
||||
"devops": "devops", "security": "security", "database": "database"}
|
||||
for tp in top_topics:
|
||||
kw = eng_kw.get(tp)
|
||||
if not kw:
|
||||
continue
|
||||
for u in c.search_users(kw, limit=10):
|
||||
login = c.login_of(u)
|
||||
if login and login not in seen_set:
|
||||
seen.append(login)
|
||||
seen_set.add(login)
|
||||
if len(seen) >= pool_cap:
|
||||
break
|
||||
return seen[:pool_cap]
|
||||
|
||||
|
||||
def profile_candidate(login: str, repo_contribs: dict[str, dict]) -> dict[str, Any]:
|
||||
"""构建候选人画像:主题向量 + 语言集合 + 活跃度 + 协作开放度。"""
|
||||
repos = c.user_repos(login, limit=15)
|
||||
texts = []
|
||||
langs: set[str] = set()
|
||||
fork_count = 0
|
||||
for r in repos:
|
||||
texts.append((r.get("description") or "") + " " + (r.get("identifier") or ""))
|
||||
if r.get("language") and isinstance(r["language"], dict):
|
||||
langs.add((r["language"].get("name") or "").lower())
|
||||
if r.get("forked_from_project_id") or r.get("forked_count"):
|
||||
fork_count += 1
|
||||
topic_vec: Counter = Counter()
|
||||
for t in texts:
|
||||
for tp in T.extract_topics(t):
|
||||
topic_vec[tp] += 1
|
||||
for lg in T.extract_languages(" ".join(texts)):
|
||||
langs.add(lg)
|
||||
|
||||
activity = 0.4
|
||||
if login in repo_contribs:
|
||||
# 本仓库贡献者 → 高活跃(贡献占比越高加成越大)
|
||||
activity = 0.7 + 0.3 * min(_parse_ratio(repo_contribs[login].get("contribution_perc")), 1.0)
|
||||
elif len(repos) >= 5:
|
||||
activity = 0.7
|
||||
elif repos:
|
||||
activity = 0.4
|
||||
collab = min(fork_count / 5.0, 1.0)
|
||||
|
||||
return {"topic_vec": dict(topic_vec), "langs": sorted(langs),
|
||||
"activity": activity, "collab": collab, "repo_count": len(repos)}
|
||||
|
||||
|
||||
def match(owner: str, repo: str, top: int = 10, pool_cap: int = 15,
|
||||
issue_sample: int = 100) -> dict[str, Any]:
|
||||
info = c.repo_info(owner, repo)
|
||||
gap_topics, needed_langs, signals = build_gap_signals(owner, repo, info, issue_sample)
|
||||
contribs_list = c.contributors(owner, repo)
|
||||
repo_contribs = {c.login_of(x): x for x in contribs_list if c.login_of(x)}
|
||||
|
||||
# 缺口主题向量(与候选人主题向量同空间)
|
||||
gap_vec = dict(gap_topics)
|
||||
|
||||
pool = candidate_pool(owner, repo, gap_topics, pool_cap)
|
||||
scored = []
|
||||
for login in pool:
|
||||
prof = profile_candidate(login, repo_contribs)
|
||||
topic_overlap = cosine(prof["topic_vec"], gap_vec)
|
||||
lang_match = jaccard(prof["langs"], needed_langs) if needed_langs else 0.0
|
||||
score = (0.45 * topic_overlap + 0.20 * lang_match
|
||||
+ 0.20 * prof["activity"] + 0.15 * prof["collab"]) * 100
|
||||
reasons: list[str] = []
|
||||
overlap_topics = sorted(set(prof["topic_vec"]) & set(gap_vec),
|
||||
key=lambda k: -prof["topic_vec"][k])
|
||||
if overlap_topics:
|
||||
reasons.append(f"覆盖缺口主题: {', '.join(overlap_topics[:4])}")
|
||||
matched_langs = sorted(set(prof["langs"]) & set(needed_langs))
|
||||
if matched_langs:
|
||||
reasons.append(f"语言匹配: {', '.join(matched_langs[:4])}")
|
||||
if login in repo_contribs:
|
||||
reasons.append("本仓库活跃贡献者")
|
||||
if prof["collab"] > 0:
|
||||
reasons.append(f"协作开放度高(fork={int(prof['collab']*5)})")
|
||||
activity_level = ("high" if prof["activity"] >= 0.7
|
||||
else "medium" if prof["activity"] >= 0.4 else "low")
|
||||
scored.append({
|
||||
"login": login, "score": round(score, 1),
|
||||
"topic_overlap": round(topic_overlap, 3),
|
||||
"language_match": round(lang_match, 3),
|
||||
"activity_level": activity_level,
|
||||
"repo_languages": prof["langs"][:6],
|
||||
"repo_count": prof["repo_count"],
|
||||
"reasons": reasons or ["无明显主题/语言重叠"],
|
||||
})
|
||||
scored.sort(key=lambda x: -x["score"])
|
||||
|
||||
top_topics = [t for t, _ in gap_topics.most_common(8)]
|
||||
return {
|
||||
"scenario": "S4_collaboration_matching",
|
||||
"repo": f"{owner}/{repo}",
|
||||
"gap_topics": top_topics,
|
||||
"needed_languages": needed_langs,
|
||||
"gap_signals": signals[:30],
|
||||
"candidates": scored[:top],
|
||||
"meta": {"pool_size": len(pool), "issue_sample": issue_sample},
|
||||
}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 渲染:Markdown 报告 + Mermaid 协作网络
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def render_report(result: dict[str, Any]) -> str:
|
||||
repo = result["repo"]
|
||||
cands = result["candidates"]
|
||||
lines = [
|
||||
f"# 科研协作智能匹配报告 — {repo}\n",
|
||||
f"> 场景 S4 · 子赛题四「应用 GitLink 辅助科研」\n",
|
||||
"## 一、仓库技术缺口分析\n",
|
||||
f"- **缺口主题**: {', '.join(result['gap_topics']) or '(未识别到明确主题)'}",
|
||||
f"- **需求语言**: {', '.join(result['needed_languages']) or '—'}",
|
||||
f"- **缺口信号样本**: {len(result['gap_signals'])} 条未解决 Issue/PR 主题证据\n",
|
||||
"| 缺口主题 | 证据(Issue/PR) | 优先级 |",
|
||||
"|----------|------------------|--------|",
|
||||
]
|
||||
for s in result["gap_signals"][:8]:
|
||||
lines.append(f"| {s['topic']} | {s['evidence']} | {s.get('priority','')} |")
|
||||
lines += ["\n## 二、推荐协作伙伴(按综合匹配分排序)\n",
|
||||
"| 排名 | 用户 | 匹配分 | 主题重叠 | 语言匹配 | 活跃度 | 匹配理由 |",
|
||||
"|------|------|--------|----------|----------|--------|----------|"]
|
||||
for i, m in enumerate(cands, 1):
|
||||
lines.append(f"| {i} | `{m['login']}` | {m['score']} | {m['topic_overlap']} | "
|
||||
f"{m['language_match']} | {m['activity_level']} | {'; '.join(m['reasons'][:2])} |")
|
||||
lines.append(f"\n_候选池规模 {result['meta']['pool_size']},issue 采样 {result['meta']['issue_sample']}_\n")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def render_mermaid(result: dict[str, Any]) -> str:
|
||||
repo = result["repo"].replace("/", "_")
|
||||
lines = ["```mermaid", "graph TD", f' R["{result["repo"]}<br/>(目标仓库)"]']
|
||||
for i, m in enumerate(result["candidates"][:8], 1):
|
||||
nid = f"C{i}"
|
||||
lines.append(f' {nid}["{m["login"]}<br/>{m["score"]}分"]')
|
||||
# 边的粗细用文字标签近似
|
||||
lines.append(f' R -- "{m["topic_overlap"]}" --> {nid}')
|
||||
lines.append("```")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description="S4 科研协作智能匹配")
|
||||
ap.add_argument("--owner", required=True)
|
||||
ap.add_argument("--repo", required=True)
|
||||
ap.add_argument("--top", type=int, default=10)
|
||||
ap.add_argument("--pool", type=int, default=15, help="候选池上限")
|
||||
ap.add_argument("--issue-sample", type=int, default=100)
|
||||
ap.add_argument("--out", help="输出目录(写 match.json/report.md/network.mmd);省略则打印 JSON")
|
||||
args = ap.parse_args()
|
||||
|
||||
result = match(args.owner, args.repo, top=args.top, pool_cap=args.pool,
|
||||
issue_sample=args.issue_sample)
|
||||
|
||||
if args.out:
|
||||
os.makedirs(args.out, exist_ok=True)
|
||||
with open(os.path.join(args.out, "match.json"), "w", encoding="utf-8") as f:
|
||||
json.dump(result, f, ensure_ascii=False, indent=2)
|
||||
with open(os.path.join(args.out, "report.md"), "w", encoding="utf-8") as f:
|
||||
f.write(render_report(result))
|
||||
with open(os.path.join(args.out, "network.mmd"), "w", encoding="utf-8") as f:
|
||||
f.write(render_mermaid(result))
|
||||
print(f"✓ S4 匹配完成 → {args.out}/match.json | report.md | network.mmd")
|
||||
print(f" 缺口主题: {', '.join(result['gap_topics'])}")
|
||||
print(f" Top 推荐: {', '.join(m['login']+'('+str(m['score'])+')' for m in result['candidates'][:5])}")
|
||||
else:
|
||||
print(json.dumps(result, ensure_ascii=False, indent=2))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
@ -0,0 +1,555 @@
|
|||
"""report.py — S5 科研进度智能跟踪与预警。
|
||||
|
||||
输入一个科研仓库,统计「本周 / 上周」的提交、Issue、PR 活跃度,结合里程碑进度,
|
||||
用阈值规则产出风险预警(stale issue / stale PR / 逾期里程碑 / 低活跃 / bus_factor),
|
||||
并给出本周相对上周的 commit 趋势。辅助科研负责人及时发现「项目停滞 / 单点风险」。
|
||||
|
||||
数据全部经 gitlink-cli 获取(commits / issue +list / pr +list / milestone +list /
|
||||
repo +contributors)。算法为纯函数、零第三方依赖,单测以 mock 数据喂入。
|
||||
|
||||
用法:
|
||||
python report.py --owner mindspore-Ecosystem --repo mindspore --out ./out
|
||||
python report.py --owner O --repo R # 仅打印 JSON
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
from collections import Counter
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from typing import Any
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
import collect as c # noqa: E402
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 常量(阈值,便于单测覆盖)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
STALE_ISSUE_DAYS = 30 # 开放且无活动 > 30 天 → stale issue
|
||||
STALE_PR_DAYS = 14 # 开放且未 review > 14 天 → stale PR
|
||||
STALE_PR_WINDOW_DAYS = 90 # 统计 stale PR 时回溯窗口(避免扫全量历史)
|
||||
LOW_ACTIVITY_COMMITS = 3 # 本周提交 < 3 → 低活跃
|
||||
BUS_FACTOR_RATIO = 0.5 # 单一贡献者占本周提交 > 50% → bus factor
|
||||
|
||||
UTC = timezone.utc
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 时间解析
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def parse_time(s: Any) -> datetime | None:
|
||||
"""把时间字段解析为带 UTC 时区的 datetime;无法解析返回 None。
|
||||
|
||||
兼容两类输入:
|
||||
- ISO 字串,如 '2024-06-01T08:30:00+08:00' / '2024-06-01T08:30:00Z'
|
||||
/ '2024-06-01 08:30:00' / '2024-06-01'
|
||||
- 整数(或整数字串)秒级 Unix 时间戳,如 1717200000 / '1717200000'
|
||||
"""
|
||||
if s is None:
|
||||
return None
|
||||
if isinstance(s, (int, float)):
|
||||
try:
|
||||
return datetime.fromtimestamp(float(s), tz=UTC)
|
||||
except (OverflowError, OSError, ValueError):
|
||||
return None
|
||||
if not isinstance(s, str):
|
||||
return None
|
||||
text = s.strip()
|
||||
if not text:
|
||||
return None
|
||||
# 纯数字 → 当作 Unix 秒级时间戳
|
||||
if text.lstrip("-").isdigit():
|
||||
try:
|
||||
return datetime.fromtimestamp(float(text), tz=UTC)
|
||||
except (OverflowError, OSError, ValueError):
|
||||
return None
|
||||
# ISO 字串:统一以 Z → +00:00
|
||||
iso = text.replace("Z", "+00:00")
|
||||
try:
|
||||
dt = datetime.fromisoformat(iso)
|
||||
except ValueError:
|
||||
# 尝试 'YYYY-MM-DD HH:MM:SS' / 'YYYY-MM-DD'
|
||||
for fmt in ("%Y-%m-%d %H:%M:%S", "%Y-%m-%d"):
|
||||
try:
|
||||
dt = datetime.strptime(text, fmt)
|
||||
break
|
||||
except ValueError:
|
||||
continue
|
||||
else:
|
||||
return None
|
||||
if dt.tzinfo is None:
|
||||
dt = dt.replace(tzinfo=UTC)
|
||||
return dt.astimezone(UTC)
|
||||
|
||||
|
||||
def in_window(dt: datetime | None, days: int, now: datetime) -> bool:
|
||||
"""判断 dt 是否落在 [now - days, now] 区间内。None 视为不在窗口内。"""
|
||||
if dt is None:
|
||||
return False
|
||||
if days < 0:
|
||||
return False
|
||||
now_utc = now.astimezone(UTC) if now.tzinfo else now.replace(tzinfo=UTC)
|
||||
dt_utc = dt.astimezone(UTC) if dt.tzinfo else dt.replace(tzinfo=UTC)
|
||||
return (now_utc - dt_utc) <= timedelta(days=days) and (now_utc - dt_utc) >= timedelta(0)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 周统计
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _commit_time(item: dict) -> datetime | None:
|
||||
for key in ("timestamp", "commit_time", "committed_date", "created_at"):
|
||||
v = item.get(key)
|
||||
dt = parse_time(v)
|
||||
if dt is not None:
|
||||
return dt
|
||||
return None
|
||||
|
||||
|
||||
def _issue_time(item: dict, prefer: tuple[str, ...]) -> datetime | None:
|
||||
for key in prefer:
|
||||
v = item.get(key)
|
||||
dt = parse_time(v)
|
||||
if dt is not None:
|
||||
return dt
|
||||
return None
|
||||
|
||||
|
||||
def _issue_status(item: dict) -> str:
|
||||
s = item.get("status")
|
||||
if isinstance(s, str) and s:
|
||||
return s.lower()
|
||||
return ""
|
||||
|
||||
|
||||
def _pr_status(item: dict) -> int:
|
||||
"""PR 状态 → 0=open,1=merged,2=closed。
|
||||
|
||||
GitLink PR 列表 status 是字符串('merged'/'open'/'closed');
|
||||
详情/health 可能是 pull_request_status 整数(0/1/2)。两者兼容。
|
||||
"""
|
||||
v = item.get("status", item.get("pull_request_status", 0))
|
||||
if isinstance(v, str):
|
||||
s = v.lower()
|
||||
if "merged" in s:
|
||||
return 1
|
||||
if s in ("closed", "close", "reject", "rejected"):
|
||||
return 2
|
||||
return 0
|
||||
try:
|
||||
return int(v)
|
||||
except (TypeError, ValueError):
|
||||
return 0
|
||||
|
||||
|
||||
def _author_login(item: dict, key: str = "author") -> str:
|
||||
"""从 commit/issue/pr 的 author 子对象取 login;退化用顶层 login/name。"""
|
||||
obj = item.get(key)
|
||||
if isinstance(obj, dict):
|
||||
for k in ("login", "name", "username"):
|
||||
if obj.get(k):
|
||||
return str(obj[k])
|
||||
for k in ("login", "name", "username", "committer_login"):
|
||||
if item.get(k):
|
||||
return str(item[k])
|
||||
return ""
|
||||
|
||||
|
||||
def _week_bounds(now: datetime) -> tuple[datetime, datetime, datetime, datetime]:
|
||||
"""返回 (this_week_start, last_week_start, last_week_end, now)(UTC)。
|
||||
|
||||
「周」按自然天对齐:本周 = [now-7d, now],上周 = [now-14d, now-7d)。
|
||||
"""
|
||||
now_utc = now.astimezone(UTC) if now.tzinfo else now.replace(tzinfo=UTC)
|
||||
this_start = now_utc - timedelta(days=7)
|
||||
last_start = now_utc - timedelta(days=14)
|
||||
last_end = this_start
|
||||
return this_start, last_start, last_end, now_utc
|
||||
|
||||
|
||||
def _commits_in(commits: list[dict], lo: datetime, hi: datetime) -> list[dict]:
|
||||
out = []
|
||||
for it in commits:
|
||||
dt = _commit_time(it)
|
||||
if dt is None:
|
||||
continue
|
||||
if lo <= dt < hi:
|
||||
out.append(it)
|
||||
return out
|
||||
|
||||
|
||||
def _window_summary(commits: list, issues: list, prs: list,
|
||||
contributors: list, lo: datetime, hi: datetime,
|
||||
now: datetime) -> dict[str, Any]:
|
||||
"""统计 [lo, hi) 区间内的活动摘要(供 week_stats 复用)。"""
|
||||
w_commits = _commits_in(commits, lo, hi)
|
||||
|
||||
opened = closed = stale_open = 0
|
||||
for iss in issues:
|
||||
created = _issue_time(iss, ("created_at", "created_unix"))
|
||||
if created is not None and lo <= created < hi:
|
||||
opened += 1
|
||||
st = _issue_status(iss)
|
||||
if st in ("closed", "reject", "rejected"):
|
||||
# 关闭时间优先 closed_at/journals_updated_at
|
||||
closed_dt = _issue_time(iss, ("closed_at", "journals_updated_at", "updated_at"))
|
||||
if closed_dt is not None and lo <= closed_dt < hi:
|
||||
closed += 1
|
||||
elif st in ("", "open", "opened"):
|
||||
# stale 判定:开放且距今 > STALE_ISSUE_DAYS 无活动
|
||||
last_dt = _issue_time(iss, ("journals_updated_at", "updated_at", "created_at"))
|
||||
if last_dt is not None and (now - last_dt) > timedelta(days=STALE_ISSUE_DAYS):
|
||||
stale_open += 1
|
||||
|
||||
pr_opened = pr_merged = pr_open_stale = 0
|
||||
for pr in prs:
|
||||
created = _issue_time(pr, ("pr_created_unix", "created_at", "created_unix"))
|
||||
if created is None:
|
||||
created = _commit_time(pr)
|
||||
if created is not None and lo <= created < hi:
|
||||
pr_opened += 1
|
||||
st = _pr_status(pr)
|
||||
if st == 1: # merged
|
||||
merged_dt = _issue_time(pr, ("pr_merged_unix", "merged_at", "updated_at"))
|
||||
if merged_dt is None:
|
||||
merged_dt = created
|
||||
if merged_dt is not None and lo <= merged_dt < hi:
|
||||
pr_merged += 1
|
||||
elif st == 0: # open
|
||||
# stale PR:开放且 created > STALE_PR_DAYS 天前、近 STALE_PR_WINDOW 天内
|
||||
if created is not None and (now - created) > timedelta(days=STALE_PR_DAYS) \
|
||||
and created > now - timedelta(days=STALE_PR_WINDOW_DAYS):
|
||||
pr_open_stale += 1
|
||||
|
||||
# 本周活跃贡献者:去重 author login(commits)
|
||||
logins = [_author_login(it) for it in w_commits]
|
||||
contrib_active = {lg for lg in logins if lg}
|
||||
return {
|
||||
"commits": len(w_commits),
|
||||
"issues_opened": opened,
|
||||
"issues_closed": closed,
|
||||
"issues_stale": stale_open,
|
||||
"prs_opened": pr_opened,
|
||||
"prs_merged": pr_merged,
|
||||
"prs_open_stale": pr_open_stale,
|
||||
"contributors_active": len(contrib_active),
|
||||
"contributors_active_logins": sorted(contrib_active),
|
||||
}
|
||||
|
||||
|
||||
def week_stats(commits: list, issues: list, prs: list,
|
||||
contributors: list, now: datetime) -> dict[str, Any]:
|
||||
"""统计本周 / 上周活跃度摘要。contributors 形参保留以兼容,活跃度以 commits 的作者为准。"""
|
||||
this_start, last_start, last_end, now_utc = _week_bounds(now)
|
||||
this = _window_summary(commits, issues, prs, contributors,
|
||||
this_start, now_utc, now_utc)
|
||||
last = _window_summary(commits, issues, prs, contributors,
|
||||
last_start, last_end, now_utc)
|
||||
return {
|
||||
"this_week": this,
|
||||
"last_week": last,
|
||||
"window": {
|
||||
"this_week_start": this_start.isoformat(),
|
||||
"now": now_utc.isoformat(),
|
||||
"last_week_start": last_start.isoformat(),
|
||||
"last_week_end": last_end.isoformat(),
|
||||
},
|
||||
"total_contributors": len(contributors),
|
||||
}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 里程碑进度
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _milestone_due(item: dict) -> datetime | None:
|
||||
for key in ("effective_date", "due_date", "deadline", "end_date"):
|
||||
dt = parse_time(item.get(key))
|
||||
if dt is not None:
|
||||
return dt
|
||||
return None
|
||||
|
||||
|
||||
def milestone_progress(milestones: list, issues: list, now: datetime) -> list[dict]:
|
||||
"""每个里程碑的 open/closed issue 数、完成率、是否逾期。
|
||||
|
||||
Issue → Milestone 的关联字段优先用 milestone_name / milestone_id。
|
||||
"""
|
||||
now_utc = now.astimezone(UTC) if now.tzinfo else now.replace(tzinfo=UTC)
|
||||
by_key: dict[Any, dict[str, int]] = {}
|
||||
for iss in issues:
|
||||
ms_name = iss.get("milestone_name") or iss.get("milestone")
|
||||
ms_id = iss.get("milestone_id") or iss.get("milestone_index")
|
||||
key = ms_name if ms_name else (ms_id if ms_id is not None else None)
|
||||
if key is None:
|
||||
continue
|
||||
bucket = by_key.setdefault(key, {"open": 0, "closed": 0})
|
||||
st = _issue_status(iss)
|
||||
if st in ("closed", "reject", "rejected"):
|
||||
bucket["closed"] += 1
|
||||
else:
|
||||
bucket["open"] += 1
|
||||
|
||||
out: list[dict] = []
|
||||
for ms in milestones:
|
||||
name = ms.get("name") or ms.get("title") or "(未命名)"
|
||||
key = name
|
||||
counts = by_key.get(key, {"open": 0, "closed": 0})
|
||||
total = counts["open"] + counts["closed"]
|
||||
pct = round(100.0 * counts["closed"] / total, 1) if total else 0.0
|
||||
due = _milestone_due(ms)
|
||||
overdue = False
|
||||
# 仅当里程碑未关闭 + 有 due_date 且 due < now → 逾期
|
||||
ms_status = str(ms.get("status", "")).lower()
|
||||
is_closed = ms_status in ("closed", "reject", "rejected", "done", "completed")
|
||||
if due is not None and not is_closed and due < now_utc:
|
||||
overdue = True
|
||||
out.append({
|
||||
"name": name,
|
||||
"open": counts["open"],
|
||||
"closed": counts["closed"],
|
||||
"total": total,
|
||||
"completion_pct": pct,
|
||||
"due_date": due.isoformat() if due else None,
|
||||
"overdue": overdue,
|
||||
"status": ms.get("status", ""),
|
||||
})
|
||||
return out
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 趋势
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def trend(this_week: dict, last_week: dict) -> dict[str, Any]:
|
||||
"""本周 vs 上周 commit 增量与活跃度等级。"""
|
||||
tc = this_week.get("commits", 0)
|
||||
lc = last_week.get("commits", 0)
|
||||
if lc == 0:
|
||||
delta_pct = 100.0 if tc > 0 else 0.0
|
||||
else:
|
||||
delta_pct = round(100.0 * (tc - lc) / lc, 1)
|
||||
if delta_pct > 10:
|
||||
level = "increasing"
|
||||
elif delta_pct < -10:
|
||||
level = "decreasing"
|
||||
else:
|
||||
level = "stable"
|
||||
return {"commit_delta_pct": delta_pct, "activity_level": level,
|
||||
"this_week_commits": tc, "last_week_commits": lc}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 风险预警
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def risk_warnings(stats: dict, milestones: list, contributors: list,
|
||||
commits: list | None = None, now: datetime | None = None) -> list[dict]:
|
||||
"""阈值规则 → 风险列表。
|
||||
|
||||
依赖 week_stats 产出的 stats(含 this_week / last_week)以及 milestone_progress
|
||||
的 milestones 列表。commits 用于 bus_factor 复算(可选,避免 stats 内无明细)。
|
||||
"""
|
||||
now_utc = (now or datetime.now(UTC)).astimezone(UTC) \
|
||||
if (now or datetime.now(UTC)).tzinfo else (now or datetime.now(UTC)).replace(tzinfo=UTC)
|
||||
this_start = now_utc - timedelta(days=7)
|
||||
warnings: list[dict] = []
|
||||
tw: dict = stats.get("this_week", {})
|
||||
|
||||
# 1) 低活跃
|
||||
commits_this = tw.get("commits", 0)
|
||||
if commits_this < LOW_ACTIVITY_COMMITS:
|
||||
warnings.append({
|
||||
"level": "warning", "type": "low_activity",
|
||||
"message": f"本周提交仅 {commits_this} 次(低于阈值 {LOW_ACTIVITY_COMMITS}),项目可能进展缓慢",
|
||||
"metric": commits_this, "suggestion": "确认是否进入收尾阶段;若无,组织一次进度同步。",
|
||||
})
|
||||
|
||||
# 2) bus_factor:单一贡献者本周提交占比 > 50%
|
||||
if commits:
|
||||
w_commits = _commits_in(commits, this_start, now_utc)
|
||||
else:
|
||||
w_commits = [] # 无明细 → 无法判 bus factor
|
||||
if w_commits:
|
||||
login_counts: Counter = Counter(_author_login(it) for it in w_commits)
|
||||
top_login, top_n = login_counts.most_common(1)[0]
|
||||
ratio = top_n / len(w_commits)
|
||||
active_n = len({lg for lg in login_counts if lg})
|
||||
if ratio > BUS_FACTOR_RATIO and active_n <= 2:
|
||||
warnings.append({
|
||||
"level": "critical", "type": "bus_factor",
|
||||
"message": f"bus factor 风险:{top_login or '(匿名)'} 一人贡献本周 {top_n}/{len(w_commits)} "
|
||||
f"次提交({round(ratio*100)}%),活跃贡献者仅 {active_n} 人",
|
||||
"metric": round(ratio, 3), "suggestion": "引入第二贡献者 / 文档化核心模块,降低单点依赖。",
|
||||
})
|
||||
|
||||
# 3) stale issue / stale PR(沿用 week_stats 已统计的口径)
|
||||
stale_iss = tw.get("issues_stale", 0)
|
||||
if stale_iss >= 5:
|
||||
warnings.append({
|
||||
"level": "warning" if stale_iss < 20 else "critical",
|
||||
"type": "stale_issue",
|
||||
"message": f"存在 {stale_iss} 个开放 Issue 超过 {STALE_ISSUE_DAYS} 天无活动",
|
||||
"metric": stale_iss, "suggestion": "分诊:关闭无效 Issue、分配负责人或拆解。",
|
||||
})
|
||||
stale_pr = tw.get("prs_open_stale", 0)
|
||||
if stale_pr >= 1:
|
||||
warnings.append({
|
||||
"level": "warning" if stale_pr < 3 else "critical",
|
||||
"type": "stale_pr",
|
||||
"message": f"存在 {stale_pr} 个开放 PR 超过 {STALE_PR_DAYS} 天未 review/合并",
|
||||
"metric": stale_pr, "suggestion": "安排 review 或明确 reject,避免 PR 堆积。",
|
||||
})
|
||||
|
||||
# 4) 逾期里程碑
|
||||
for ms in milestones:
|
||||
if isinstance(ms, dict) and ms.get("overdue"):
|
||||
warnings.append({
|
||||
"level": "critical", "type": "overdue_milestone",
|
||||
"message": f"里程碑「{ms.get('name', '?')}」已逾期"
|
||||
+ (f"(due {ms.get('due_date')})" if ms.get("due_date") else "")
|
||||
+ f",完成率 {ms.get('completion_pct', 0)}%",
|
||||
"metric": ms.get("completion_pct", 0),
|
||||
"suggestion": "重新评估范围或顺延 deadline,并同步干系人。",
|
||||
})
|
||||
|
||||
# 排序:critical > warning > info
|
||||
rank = {"critical": 0, "warning": 1, "info": 2}
|
||||
warnings.sort(key=lambda w: (rank.get(w["level"], 9), w["type"]))
|
||||
return warnings
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 主入口(取数 + 算法)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def analyze(owner: str, repo: str, now: datetime | None = None) -> dict[str, Any]:
|
||||
"""取数 + 算法:返回完整结果 dict。"""
|
||||
if now is None:
|
||||
now = datetime.now(UTC)
|
||||
commits = c.commits(owner, repo, max_pages=10)
|
||||
issues = c.issues_all(owner, repo)
|
||||
prs = c.prs_all(owner, repo)
|
||||
milestones = c.milestones(owner, repo)
|
||||
contributors = c.contributors(owner, repo)
|
||||
|
||||
stats = week_stats(commits, issues, prs, contributors, now)
|
||||
ms_progress = milestone_progress(milestones, issues, now)
|
||||
warnings = risk_warnings(stats, ms_progress, contributors, commits=commits, now=now)
|
||||
tr = trend(stats["this_week"], stats["last_week"])
|
||||
|
||||
return {
|
||||
"scenario": "S5_progress_tracking",
|
||||
"repo": f"{owner}/{repo}",
|
||||
"generated_at": now.astimezone(UTC).isoformat(),
|
||||
"week_stats": stats,
|
||||
"trend": tr,
|
||||
"milestones": ms_progress,
|
||||
"risk_warnings": warnings,
|
||||
"meta": {
|
||||
"commits_fetched": len(commits),
|
||||
"issues_fetched": len(issues),
|
||||
"prs_fetched": len(prs),
|
||||
"milestones_fetched": len(milestones),
|
||||
"contributors_fetched": len(contributors),
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 渲染:Markdown 周报
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def render_report(result: dict[str, Any]) -> str:
|
||||
repo = result["repo"]
|
||||
stats = result["week_stats"]
|
||||
tw, lw = stats["this_week"], stats["last_week"]
|
||||
tr = result["trend"]
|
||||
lines = [
|
||||
f"# 科研进度智能跟踪周报 — {repo}\n",
|
||||
f"> 场景 S5 · 子赛题四「应用 GitLink 辅助科研」· 生成于 {result['generated_at']}\n",
|
||||
"## 一、本周 / 上周活动对比\n",
|
||||
"| 指标 | 本周 | 上周 |",
|
||||
"|------|------|------|",
|
||||
f"| 提交 commits | {tw['commits']} | {lw['commits']} |",
|
||||
f"| Issue 新增 | {tw['issues_opened']} | {lw['issues_opened']} |",
|
||||
f"| Issue 关闭 | {tw['issues_closed']} | {lw['issues_closed']} |",
|
||||
f"| 开放 stale issue (>{STALE_ISSUE_DAYS}天) | {tw['issues_stale']} | {lw['issues_stale']} |",
|
||||
f"| PR 新增 | {tw['prs_opened']} | {lw['prs_opened']} |",
|
||||
f"| PR 合并 | {tw['prs_merged']} | {lw['prs_merged']} |",
|
||||
f"| 开放 stale PR (>{STALE_PR_DAYS}天) | {tw['prs_open_stale']} | {lw['prs_open_stale']} |",
|
||||
f"| 活跃贡献者 | {tw['contributors_active']} | {lw['contributors_active']} |",
|
||||
"",
|
||||
f"- **趋势**:commit 周环比 **{tr['commit_delta_pct']}%**,活跃度等级 `{tr['activity_level']}`",
|
||||
"",
|
||||
"## 二、里程碑进度\n",
|
||||
]
|
||||
ms = result["milestones"]
|
||||
if ms:
|
||||
lines += [
|
||||
"| 里程碑 | 完成/总数 | 完成率 | due_date | 状态 |",
|
||||
"|--------|-----------|--------|----------|------|",
|
||||
]
|
||||
for m in ms:
|
||||
flag = " ⚠️逾期" if m["overdue"] else ""
|
||||
lines.append(
|
||||
f"| {m['name']}{flag} | {m['closed']}/{m['total']} | {m['completion_pct']}% "
|
||||
f"| {m['due_date'] or '—'} | {m['status'] or '—'} |"
|
||||
)
|
||||
else:
|
||||
lines.append("_(仓库无里程碑数据)_")
|
||||
|
||||
lines += ["\n## 三、风险预警\n"]
|
||||
warns = result["risk_warnings"]
|
||||
if warns:
|
||||
lines += ["| 级别 | 类型 | 说明 | 建议 |", "|------|------|------|------|"]
|
||||
for w in warns:
|
||||
lines.append(f"| {w['level']} | {w['type']} | {w['message']} | {w['suggestion']} |")
|
||||
else:
|
||||
lines.append("_(未触发风险阈值,进度正常)_")
|
||||
|
||||
lines += [
|
||||
"\n## 四、附\n",
|
||||
f"- 取数:commits={result['meta']['commits_fetched']} "
|
||||
f"issues={result['meta']['issues_fetched']} prs={result['meta']['prs_fetched']} "
|
||||
f"milestones={result['meta']['milestones_fetched']} "
|
||||
f"contributors={result['meta']['contributors_fetched']}",
|
||||
f"- 窗口:本周 [{stats['window']['this_week_start']}, {stats['window']['now']}];"
|
||||
f"上周 [{stats['window']['last_week_start']}, {stats['window']['last_week_end']})",
|
||||
f"- 阈值:stale_issue>{STALE_ISSUE_DAYS}天 / stale_pr>{STALE_PR_DAYS}天 / "
|
||||
f"低活跃<{LOW_ACTIVITY_COMMITS}次/周 / bus_factor>{int(BUS_FACTOR_RATIO*100)}%",
|
||||
"",
|
||||
]
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description="S5 科研进度智能跟踪与预警")
|
||||
ap.add_argument("--owner", required=True)
|
||||
ap.add_argument("--repo", required=True)
|
||||
ap.add_argument("--out", help="输出目录(写 report.json + weekly_report.md);省略则打印 JSON")
|
||||
args = ap.parse_args()
|
||||
|
||||
result = analyze(args.owner, args.repo)
|
||||
|
||||
if args.out:
|
||||
os.makedirs(args.out, exist_ok=True)
|
||||
with open(os.path.join(args.out, "report.json"), "w", encoding="utf-8") as f:
|
||||
json.dump(result, f, ensure_ascii=False, indent=2)
|
||||
with open(os.path.join(args.out, "weekly_report.md"), "w", encoding="utf-8") as f:
|
||||
f.write(render_report(result))
|
||||
print(f"✓ S5 周报完成 → {args.out}/report.json | weekly_report.md")
|
||||
print(f" 本周提交 {result['week_stats']['this_week']['commits']} "
|
||||
f"(趋势 {result['trend']['activity_level']});风险 {len(result['risk_warnings'])} 条")
|
||||
else:
|
||||
print(json.dumps(result, ensure_ascii=False, indent=2))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
@ -0,0 +1,573 @@
|
|||
"""repro.py — S3 科研项目合规与复现性检查。
|
||||
|
||||
输入一个科研仓库,检查其「合规性」(许可证、版权、依赖、安全策略、数据隐私)
|
||||
与「可复现性」(CI 配置、lockfile、README 是否含数据集/环境/构建说明、版本 tag、
|
||||
密钥泄露),分别给出 0-10 的复现分与合规分,并产出检查清单、风险项与中文报告。
|
||||
|
||||
数据全部经 gitlink-cli 获取:
|
||||
- repo +info(仓库信息、版本 tag)
|
||||
- file +get(LICENSE / README / go.mod / requirements.txt / package.json / .gitignore 等)
|
||||
- repo +tree(扫 data/、.env、config、.gitea/.github workflows 等是否存在)
|
||||
|
||||
用法:
|
||||
python repro.py --owner mindspore-Ecosystem --repo mindspore --out ./out
|
||||
python repro.py --owner O --repo R # 仅打印 JSON
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from typing import Any
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
import collect as c # noqa: E402
|
||||
import gitlink_data as gd # noqa: E402
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 常量
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# 复现性/合规相关的关键文件(相对仓库根)
|
||||
KEY_FILES: tuple[str, ...] = (
|
||||
"LICENSE", "LICENSE.txt", "LICENSE.md",
|
||||
"README", "README.md", "README.rst",
|
||||
"go.mod", "requirements.txt", "package.json", "Cargo.toml",
|
||||
".gitignore", "SECURITY.md", "CONTRIBUTING.md",
|
||||
)
|
||||
|
||||
# CI 配置目录/文件(复现性信号)。
|
||||
# 注意:根 tree 通常只列顶层目录(.github/.devops/.gitea),不一定展开到 workflows/,
|
||||
# 故同时收录顶层目录名与深层路径;.devops 是 GitLink 专属 CI/CD 目录。
|
||||
CI_PATHS: tuple[str, ...] = (
|
||||
".devops",
|
||||
".gitea", ".gitea/workflows",
|
||||
".github", ".github/workflows",
|
||||
".gitlab-ci.yml", ".circleci", ".travis.yml", "azure-pipelines.yml",
|
||||
)
|
||||
|
||||
# 锁文件(复现性信号:依赖版本固定)
|
||||
LOCKFILES: tuple[str, ...] = (
|
||||
"go.sum", "package-lock.json", "yarn.lock", "pnpm-lock.yaml",
|
||||
"requirements.txt", "poetry.lock", "Pipfile.lock", "Cargo.lock", "composer.lock",
|
||||
)
|
||||
|
||||
# 敏感目录/文件(数据隐私信号)
|
||||
SENSITIVE_PATHS: tuple[str, ...] = (
|
||||
"data/", "data", "dataset/", "datasets/", ".env", ".env.local",
|
||||
"config/secrets", "secrets",
|
||||
)
|
||||
|
||||
# README 复现性关键词(数据集 / 环境 / 构建 / 运行说明)
|
||||
README_REPRO_KEYWORDS: tuple[str, ...] = (
|
||||
"install", "setup", "环境", "依赖", "build", "构建", "运行", "run",
|
||||
"dataset", "数据集", "docker", "conda", "pip install", "npm install",
|
||||
"requirements", "reproduce", "复现", "环境配置", "usage", "用法",
|
||||
)
|
||||
|
||||
# 许可证识别关键词(顺序即优先级)
|
||||
LICENSE_PATTERNS: tuple[tuple[str, str], ...] = (
|
||||
("MulanPSL", "MulanPSL-2.0"),
|
||||
("木兰宽松许可证", "MulanPSL-2.0"),
|
||||
("Apache License", "Apache-2.0"),
|
||||
("MIT License", "MIT"),
|
||||
("GNU GENERAL PUBLIC LICENSE", "GPL"),
|
||||
("GNU Lesser General Public License", "LGPL"),
|
||||
("BSD ", "BSD"),
|
||||
("ISC License", "ISC"),
|
||||
("Mozilla Public License", "MPL"),
|
||||
("Unlicense", "Unlicense"),
|
||||
)
|
||||
|
||||
# 密钥/敏感信息正则(按类别)
|
||||
SECRET_PATTERNS: tuple[tuple[str, str, str], ...] = (
|
||||
# (category, level, regex)
|
||||
("private_key", "critical", r"-----BEGIN (?:RSA |EC |DSA |OPENSSH |PGP )?PRIVATE KEY-----"),
|
||||
("aws_access_key", "critical", r"AKIA[0-9A-Z]{16}"),
|
||||
("aws_secret", "critical", r"aws_secret_access_key\s*[:=]\s*['\"]?[A-Za-z0-9/+=]{40}"),
|
||||
("generic_api_key", "high", r"(?i)api[_-]?key\s*[:=]\s*['\"]?[A-Za-z0-9_\-]{16,}"),
|
||||
("google_api_key", "high", r"AIza[0-9A-Za-z_\-]{35}"),
|
||||
("slack_token", "high", r"xox[baprs]-[0-9A-Za-z-]{10,}"),
|
||||
("github_token", "high", r"gh[pousr]_[A-Za-z0-9]{36,}"),
|
||||
("jwt", "medium", r"eyJ[A-Za-z0-9_\-]+\.eyJ[A-Za-z0-9_\-]+\.[A-Za-z0-9_\-]+"),
|
||||
("email", "low", r"[a-zA-Z0-9._%+\-]+@[a-zA-Z0-9.\-]+\.[A-Za-z]{2,}"),
|
||||
# 中国大陆手机号
|
||||
("phone_cn", "low", r"(?<!\d)1[3-9]\d{9}(?!\d)"),
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 算法:纯函数(单测对象,不联网)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def identify_license(text: str) -> dict[str, Any]:
|
||||
"""关键词匹配 LICENSE 文本,返回许可证信息。
|
||||
|
||||
返回: {"license": str, "recognized": bool, "evidence": str}
|
||||
未识别返回 license="None"。
|
||||
"""
|
||||
if not text or not text.strip():
|
||||
return {"license": "None", "recognized": False,
|
||||
"evidence": "LICENSE 文件为空或缺失"}
|
||||
low = text.lower()
|
||||
for pat, name in LICENSE_PATTERNS:
|
||||
if pat.lower() in low:
|
||||
# 找到关键词所在行作为证据
|
||||
idx = low.find(pat.lower())
|
||||
line_start = text.rfind("\n", 0, idx) + 1
|
||||
line_end = text.find("\n", idx)
|
||||
if line_end == -1:
|
||||
line_end = len(text)
|
||||
evidence = text[line_start:line_end].strip()[:120]
|
||||
return {"license": name, "recognized": True, "evidence": evidence}
|
||||
return {"license": "None", "recognized": False,
|
||||
"evidence": "未匹配到已知许可证关键词"}
|
||||
|
||||
|
||||
def scan_secrets(text: str, file: str = "") -> list[dict[str, Any]]:
|
||||
"""扫描文本中的密钥/敏感信息。
|
||||
|
||||
返回: [{"level","category","file","line","detail"}, ...]
|
||||
"""
|
||||
if not text:
|
||||
return []
|
||||
findings: list[dict[str, Any]] = []
|
||||
lines = text.splitlines()
|
||||
for category, level, pattern in SECRET_PATTERNS:
|
||||
for m in re.finditer(pattern, text):
|
||||
# 计算行号与所在行内容
|
||||
line_no = text.count("\n", 0, m.start()) + 1
|
||||
line_content = lines[line_no - 1] if 0 <= line_no - 1 < len(lines) else ""
|
||||
detail = m.group(0)
|
||||
# 脱敏:长串截断
|
||||
if len(detail) > 40:
|
||||
detail = detail[:20] + "..." + detail[-6:]
|
||||
findings.append({
|
||||
"level": level, "category": category, "file": file,
|
||||
"line": line_no, "detail": detail,
|
||||
"context": line_content.strip()[:80],
|
||||
})
|
||||
return findings
|
||||
|
||||
|
||||
def _tree_paths(tree: list) -> list[str]:
|
||||
"""从 tree 列表提取所有路径字符串(兼容多种字段名)。"""
|
||||
paths: list[str] = []
|
||||
for item in tree:
|
||||
if isinstance(item, dict):
|
||||
for k in ("path", "name", "filepath"):
|
||||
v = item.get(k)
|
||||
if isinstance(v, str) and v:
|
||||
paths.append(v)
|
||||
break
|
||||
elif isinstance(item, str) and item:
|
||||
paths.append(item)
|
||||
return paths
|
||||
|
||||
|
||||
def _has_path(paths: list[str], targets: tuple[str, ...]) -> list[str]:
|
||||
"""paths 中命中任一 target(前缀/精确都算)的命中项,返回命中原文(去重保序)。"""
|
||||
hits: list[str] = []
|
||||
seen: set[str] = set()
|
||||
low_targets = [t.lower().rstrip("/") for t in targets]
|
||||
for p in paths:
|
||||
pl = p.lower().rstrip("/")
|
||||
for tl in low_targets:
|
||||
if pl == tl or pl.startswith(tl + "/"):
|
||||
key = f"{p}::{tl}"
|
||||
if key not in seen:
|
||||
seen.add(key)
|
||||
hits.append(p)
|
||||
break
|
||||
return hits
|
||||
|
||||
|
||||
def repro_checks(file_texts: dict[str, str], tree: list,
|
||||
repo_info: dict | None = None) -> list[dict[str, Any]]:
|
||||
"""复现性检查。
|
||||
|
||||
file_texts: {路径: 文本内容}(已取好)
|
||||
tree: tree 列表(已取好)
|
||||
repo_info: 仓库信息(取版本 tag;无 tag 字段则 unknown)
|
||||
|
||||
返回检查项列表: [{"name","pass","score","evidence"}]
|
||||
每项 score 0-2(0=缺失/失败, 1=部分, 2=完备)。
|
||||
"""
|
||||
paths = _tree_paths(tree)
|
||||
items: list[dict[str, Any]] = []
|
||||
|
||||
# 1) CI 配置
|
||||
ci_hits = _has_path(paths, CI_PATHS)
|
||||
if ci_hits:
|
||||
items.append({"name": "CI 配置", "pass": True, "score": 2,
|
||||
"evidence": f"检测到 CI 配置: {', '.join(ci_hits[:3])}"})
|
||||
else:
|
||||
items.append({"name": "CI 配置", "pass": False, "score": 0,
|
||||
"evidence": "未找到 .gitea/.github/.gitlab 等 CI 配置"})
|
||||
|
||||
# 2) lockfile(依赖版本固定)
|
||||
lock_hits = _has_path(paths, LOCKFILES)
|
||||
if lock_hits:
|
||||
items.append({"name": "依赖锁文件", "pass": True, "score": 2,
|
||||
"evidence": f"存在 lockfile: {', '.join(lock_hits[:3])}"})
|
||||
else:
|
||||
items.append({"name": "依赖锁文件", "pass": False, "score": 0,
|
||||
"evidence": "未找到 go.sum/package-lock.json/poetry.lock 等锁文件"})
|
||||
|
||||
# 3) README 含数据集/环境/构建说明
|
||||
readme_text = ""
|
||||
for k in ("README.md", "README", "README.rst"):
|
||||
if k in file_texts and file_texts[k]:
|
||||
readme_text = file_texts[k]
|
||||
break
|
||||
if readme_text:
|
||||
low = readme_text.lower()
|
||||
hit_kw = [kw for kw in README_REPRO_KEYWORDS if kw.lower() in low]
|
||||
score = 2 if len(hit_kw) >= 4 else (1 if len(hit_kw) >= 1 else 0)
|
||||
items.append({"name": "README 复现说明", "pass": score > 0, "score": score,
|
||||
"evidence": (f"README 含复现关键词 {len(hit_kw)} 个: {', '.join(hit_kw[:5])}"
|
||||
if hit_kw else "README 存在但缺少数据集/环境/构建说明")})
|
||||
else:
|
||||
items.append({"name": "README 复现说明", "pass": False, "score": 0,
|
||||
"evidence": "未找到 README"})
|
||||
|
||||
# 4) 版本 tag(用 repo_info,无 tag 字段则 unknown)
|
||||
info = repo_info or {}
|
||||
tag = (info.get("version") or info.get("tag") or info.get("release_tag")
|
||||
or info.get("default_branch") or "")
|
||||
# 多数 GitLink repo_info 无显式 tag 字段 → 标 unknown(不扣分但提示)
|
||||
has_tag = bool(info.get("version") or info.get("tag") or info.get("release_tag"))
|
||||
if has_tag:
|
||||
items.append({"name": "版本 tag", "pass": True, "score": 2,
|
||||
"evidence": f"版本/release tag: {tag}"})
|
||||
else:
|
||||
items.append({"name": "版本 tag", "pass": False, "score": 1,
|
||||
"evidence": f"repo_info 无显式 tag 字段(默认分支: {info.get('default_branch', 'unknown')}),建议打 tag 固定可复现版本"})
|
||||
|
||||
# 5) 容器化(Dockerfile / docker-compose)—— 复现环境
|
||||
container_hits = _has_path(paths, ("Dockerfile", "docker-compose.yml",
|
||||
"docker-compose.yaml", ".devcontainer"))
|
||||
if container_hits:
|
||||
items.append({"name": "容器化环境", "pass": True, "score": 2,
|
||||
"evidence": f"存在容器配置: {', '.join(container_hits[:3])}"})
|
||||
else:
|
||||
items.append({"name": "容器化环境", "pass": False, "score": 0,
|
||||
"evidence": "未找到 Dockerfile/docker-compose,复现环境依赖手工描述"})
|
||||
|
||||
return items
|
||||
|
||||
|
||||
def compliance_items(license_info: dict[str, Any], file_texts: dict[str, str],
|
||||
tree: list) -> list[dict[str, Any]]:
|
||||
"""合规性检查。
|
||||
|
||||
返回检查项列表: [{"name","pass","score","evidence"}],score 0-2。
|
||||
"""
|
||||
paths = _tree_paths(tree)
|
||||
items: list[dict[str, Any]] = []
|
||||
|
||||
# 1) LICENSE 声明
|
||||
lic = license_info.get("license", "None")
|
||||
recognized = license_info.get("recognized", False)
|
||||
if recognized and lic != "None":
|
||||
items.append({"name": "LICENSE 文件", "pass": True, "score": 2,
|
||||
"evidence": f"LICENSE 声明为 {lic}"})
|
||||
elif lic == "None" and not (file_texts.get("LICENSE") or file_texts.get("LICENSE.txt")
|
||||
or file_texts.get("LICENSE.md")):
|
||||
items.append({"name": "LICENSE 文件", "pass": False, "score": 0,
|
||||
"evidence": "缺少 LICENSE 文件"})
|
||||
else:
|
||||
items.append({"name": "LICENSE 文件", "pass": False, "score": 1,
|
||||
"evidence": "LICENSE 文件存在但类型未识别"})
|
||||
|
||||
# 2) SECURITY.md
|
||||
sec_hits = _has_path(paths, ("SECURITY.md", "SECURITY", "security.md"))
|
||||
if sec_hits:
|
||||
items.append({"name": "安全策略 SECURITY.md", "pass": True, "score": 2,
|
||||
"evidence": f"存在 {sec_hits[0]}"})
|
||||
else:
|
||||
items.append({"name": "安全策略 SECURITY.md", "pass": False, "score": 0,
|
||||
"evidence": "缺少 SECURITY.md,无安全披露流程"})
|
||||
|
||||
# 3) 版权头(采样 README/LICENSE 头部判断有无 Copyright)
|
||||
sample = (file_texts.get("LICENSE", "") + "\n" + file_texts.get("README.md", "")
|
||||
+ "\n" + file_texts.get("README", ""))
|
||||
has_copyright = ("copyright" in sample.lower()) or ("版权" in sample) or ("©" in sample)
|
||||
if has_copyright:
|
||||
items.append({"name": "版权声明", "pass": True, "score": 2,
|
||||
"evidence": "LICENSE/README 中含 copyright/版权 声明"})
|
||||
else:
|
||||
items.append({"name": "版权声明", "pass": False, "score": 1,
|
||||
"evidence": "未在 LICENSE/README 中发现版权声明(建议源文件头补 Copyright 注释)"})
|
||||
|
||||
# 4) 依赖合规(存在依赖清单即视为已声明,识别许可证更佳)
|
||||
dep_present = bool(_has_path(paths, ("go.mod", "requirements.txt", "package.json",
|
||||
"Cargo.toml", "pom.xml", "setup.py", "pyproject.toml")))
|
||||
if dep_present:
|
||||
items.append({"name": "依赖清单声明", "pass": True, "score": 2,
|
||||
"evidence": "存在依赖管理文件(建议核对各依赖许可证兼容性)"})
|
||||
else:
|
||||
items.append({"name": "依赖清单声明", "pass": False, "score": 1,
|
||||
"evidence": "未发现标准依赖管理文件"})
|
||||
|
||||
# 5) CONTRIBUTING.md(社区合规)
|
||||
contrib_hits = _has_path(paths, ("CONTRIBUTING.md", "CONTRIBUTING", "contributing.md"))
|
||||
if contrib_hits:
|
||||
items.append({"name": "贡献指南", "pass": True, "score": 2,
|
||||
"evidence": f"存在 {contrib_hits[0]}"})
|
||||
else:
|
||||
items.append({"name": "贡献指南", "pass": False, "score": 1,
|
||||
"evidence": "缺少 CONTRIBUTING.md"})
|
||||
|
||||
return items
|
||||
|
||||
|
||||
def data_privacy(tree: list, gitignore_text: str) -> dict[str, Any]:
|
||||
"""数据隐私检查。
|
||||
|
||||
返回: {
|
||||
"items": [{"name","pass","score","evidence"}],
|
||||
"risks": [...], # 高风险项明细
|
||||
}
|
||||
"""
|
||||
paths = _tree_paths(tree)
|
||||
items: list[dict[str, Any]] = []
|
||||
risks: list[str] = []
|
||||
|
||||
# 1) data/ 目录是否入库
|
||||
data_hits = _has_path(paths, ("data/", "dataset/", "datasets/"))
|
||||
if data_hits:
|
||||
items.append({"name": "数据目录入库", "pass": False, "score": 0,
|
||||
"evidence": f"data/ 目录已入库: {', '.join(data_hits[:3])}(建议大文件走外部存储/DVC)"})
|
||||
risks.append(f"数据目录入库: {', '.join(data_hits[:3])}(可能含敏感数据)")
|
||||
else:
|
||||
items.append({"name": "数据目录入库", "pass": True, "score": 2,
|
||||
"evidence": "未发现 data/ 目录入库"})
|
||||
|
||||
# 2) .env 是否入库
|
||||
env_hits = _has_path(paths, (".env", ".env.local", ".env.production"))
|
||||
if env_hits:
|
||||
items.append({"name": ".env 入库", "pass": False, "score": 0,
|
||||
"evidence": f".env 已入库: {', '.join(env_hits[:3])}(高风险,疑似凭据泄露)"})
|
||||
risks.append(f".env 已入库: {', '.join(env_hits[:3])}(凭据泄露风险)")
|
||||
else:
|
||||
items.append({"name": ".env 入库", "pass": True, "score": 2,
|
||||
"evidence": ".env 未入库"})
|
||||
|
||||
# 3) .gitignore 是否忽略 .env
|
||||
gi = (gitignore_text or "").lower()
|
||||
ignores_env = ".env" in gi
|
||||
if ignores_env:
|
||||
items.append({"name": ".gitignore 忽略 .env", "pass": True, "score": 2,
|
||||
"evidence": ".gitignore 已配置忽略 .env"})
|
||||
else:
|
||||
items.append({"name": ".gitignore 忽略 .env", "pass": False, "score": 1,
|
||||
"evidence": ".gitignore 未忽略 .env(建议添加 .env)"})
|
||||
if not env_hits:
|
||||
risks.append(".gitignore 未忽略 .env(预防性建议)")
|
||||
|
||||
return {"items": items, "risks": risks}
|
||||
|
||||
|
||||
def _score_10(items: list[dict[str, Any]], cap: float = 10.0) -> float:
|
||||
"""把检查项的 0-2 分聚合为 0-10 分:sum(score)/sum(max=2) * 10。"""
|
||||
total = sum(it.get("score", 0) for it in items)
|
||||
max_total = sum(2 for _ in items)
|
||||
if max_total == 0:
|
||||
return 0.0
|
||||
return round(min(cap, total / max_total * cap), 1)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 数据采集(调 gitlink-cli;算法不依赖本节)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def collect_file_texts(owner: str, repo: str, ref: str = "master") -> dict[str, str]:
|
||||
"""批量取关键文件文本。缺失文件返回空串(不出现在 dict 中)。"""
|
||||
out: dict[str, str] = {}
|
||||
for path in KEY_FILES:
|
||||
try:
|
||||
txt = c.file_text(owner, repo, path, ref=ref)
|
||||
except Exception:
|
||||
txt = ""
|
||||
if txt and txt.strip():
|
||||
out[path] = txt
|
||||
return out
|
||||
|
||||
|
||||
def collect_tree(owner: str, repo: str, ref: str = "master") -> list:
|
||||
"""取根 tree(含扫 data/、.env、config、workflows 等)。"""
|
||||
try:
|
||||
return c.tree(owner, repo, path="", ref=ref)
|
||||
except Exception:
|
||||
return []
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 主流程
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def run(owner: str, repo: str) -> dict[str, Any]:
|
||||
info = c.repo_info(owner, repo)
|
||||
ref = info.get("default_branch") or "master"
|
||||
|
||||
file_texts = collect_file_texts(owner, repo, ref)
|
||||
tree = collect_tree(owner, repo, ref)
|
||||
|
||||
license_text = (file_texts.get("LICENSE") or file_texts.get("LICENSE.txt")
|
||||
or file_texts.get("LICENSE.md") or "")
|
||||
license_info = identify_license(license_text)
|
||||
|
||||
repro = repro_checks(file_texts, tree, repo_info=info)
|
||||
compliance = compliance_items(license_info, file_texts, tree)
|
||||
gitignore_text = file_texts.get(".gitignore", "")
|
||||
dp = data_privacy(tree, gitignore_text)
|
||||
|
||||
# 汇总密钥扫描(扫所有已取文本 + gitignore)
|
||||
all_secrets: list[dict[str, Any]] = []
|
||||
for path, txt in file_texts.items():
|
||||
all_secrets.extend(scan_secrets(txt, file=path))
|
||||
|
||||
repro_score = _score_10(repro)
|
||||
compliance_score = _score_10(compliance + dp["items"])
|
||||
|
||||
# 风险项汇总
|
||||
risks: list[dict[str, Any]] = []
|
||||
for it in repro + compliance + dp["items"]:
|
||||
if not it["pass"]:
|
||||
risks.append({"area": "repro/compliance", "name": it["name"],
|
||||
"evidence": it["evidence"], "level": "medium"})
|
||||
for r in dp["risks"]:
|
||||
risks.append({"area": "privacy", "name": "数据隐私", "evidence": r, "level": "high"})
|
||||
for s in all_secrets:
|
||||
risks.append({"area": "secret", "name": s["category"], "file": s["file"],
|
||||
"line": s["line"], "detail": s["detail"],
|
||||
"level": s["level"]})
|
||||
# 按级别排序
|
||||
level_rank = {"critical": 0, "high": 1, "medium": 2, "low": 3}
|
||||
risks.sort(key=lambda r: level_rank.get(r.get("level", "low"), 9))
|
||||
|
||||
return {
|
||||
"scenario": "S3_compliance_reproducibility",
|
||||
"repo": f"{owner}/{repo}",
|
||||
"default_branch": ref,
|
||||
"license": license_info["license"],
|
||||
"repro_items": repro,
|
||||
"compliance_items": compliance,
|
||||
"privacy_items": dp["items"],
|
||||
"secrets": all_secrets,
|
||||
"risks": risks,
|
||||
"repro_score": repro_score,
|
||||
"compliance_score": compliance_score,
|
||||
"meta": {"key_files_found": sorted(file_texts.keys()),
|
||||
"tree_size": len(tree),
|
||||
"languages": c.languages(owner, repo)},
|
||||
}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 渲染:Markdown 报告
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _grade(score: float) -> str:
|
||||
if score >= 8:
|
||||
return "良好"
|
||||
if score >= 6:
|
||||
return "及格"
|
||||
if score >= 4:
|
||||
return "偏弱"
|
||||
return "较差"
|
||||
|
||||
|
||||
def render_report(result: dict[str, Any]) -> str:
|
||||
repo = result["repo"]
|
||||
lic = result["license"]
|
||||
rs = result["repro_score"]
|
||||
cs = result["compliance_score"]
|
||||
lines = [
|
||||
f"# 科研项目合规与复现性检查报告 — {repo}\n",
|
||||
f"> 场景 S3 · 子赛题四「应用 GitLink 辅助科研」\n",
|
||||
f"- **默认分支**: `{result.get('default_branch', 'master')}`",
|
||||
f"- **识别许可证**: `{lic}`",
|
||||
f"- **复现性评分**: **{rs}/10**({_grade(rs)})",
|
||||
f"- **合规性评分**: **{cs}/10**({_grade(cs)})\n",
|
||||
]
|
||||
|
||||
# 检查清单表
|
||||
lines += ["## 一、复现性检查清单\n",
|
||||
"| 检查项 | 通过 | 得分 | 证据 |",
|
||||
"|--------|:----:|:----:|------|"]
|
||||
for it in result["repro_items"]:
|
||||
mark = "PASS" if it["pass"] else "FAIL"
|
||||
lines.append(f"| {it['name']} | {mark} | {it['score']}/2 | {it['evidence']} |")
|
||||
|
||||
lines += ["\n## 二、合规性检查清单\n",
|
||||
"| 检查项 | 通过 | 得分 | 证据 |",
|
||||
"|--------|:----:|:----:|------|"]
|
||||
for it in result["compliance_items"]:
|
||||
mark = "PASS" if it["pass"] else "FAIL"
|
||||
lines.append(f"| {it['name']} | {mark} | {it['score']}/2 | {it['evidence']} |")
|
||||
|
||||
lines += ["\n## 三、数据隐私检查\n",
|
||||
"| 检查项 | 通过 | 得分 | 证据 |",
|
||||
"|--------|:----:|:----:|------|"]
|
||||
for it in result["privacy_items"]:
|
||||
mark = "PASS" if it["pass"] else "FAIL"
|
||||
lines.append(f"| {it['name']} | {mark} | {it['score']}/2 | {it['evidence']} |")
|
||||
|
||||
# 风险项表
|
||||
risks = result["risks"]
|
||||
lines += ["\n## 四、风险项(按严重程度排序)\n",
|
||||
"| 级别 | 类别 | 名称 | 文件:行 | 证据 |",
|
||||
"|:----:|------|------|---------|------|"]
|
||||
if risks:
|
||||
for r in risks:
|
||||
lvl = r.get("level", "medium")
|
||||
area = r.get("area", "")
|
||||
name = r.get("name", "")
|
||||
file_loc = ""
|
||||
if r.get("file"):
|
||||
file_loc = f"{r['file']}:{r.get('line', '')}"
|
||||
ev = r.get("evidence") or r.get("detail", "")
|
||||
lines.append(f"| {lvl} | {area} | {name} | {file_loc} | {ev} |")
|
||||
else:
|
||||
lines.append("| — | — | 无风险项 | — | 全部检查通过 |")
|
||||
|
||||
# 密钥小结
|
||||
secrets = result.get("secrets", [])
|
||||
if secrets:
|
||||
lines.append(f"\n> 检出 **{len(secrets)}** 处疑似敏感信息(见风险项表),请人工复核确认。\n")
|
||||
|
||||
lines.append(f"\n_复现分 {rs}/10 · 合规分 {cs}/10 · 树节点 {result.get('meta',{}).get('tree_size','?')}_\n")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description="S3 科研项目合规与复现性检查")
|
||||
ap.add_argument("--owner", required=True)
|
||||
ap.add_argument("--repo", required=True)
|
||||
ap.add_argument("--out", help="输出目录(写 repro.json + compliance_report.md);省略则打印 JSON")
|
||||
args = ap.parse_args()
|
||||
|
||||
result = run(args.owner, args.repo)
|
||||
|
||||
if args.out:
|
||||
os.makedirs(args.out, exist_ok=True)
|
||||
with open(os.path.join(args.out, "repro.json"), "w", encoding="utf-8") as f:
|
||||
json.dump(result, f, ensure_ascii=False, indent=2)
|
||||
with open(os.path.join(args.out, "compliance_report.md"), "w", encoding="utf-8") as f:
|
||||
f.write(render_report(result))
|
||||
print(f"✓ S3 合规/复现检查完成 → {args.out}/repro.json | compliance_report.md")
|
||||
print(f" 许可证: {result['license']}")
|
||||
print(f" 复现分: {result['repro_score']}/10 合规分: {result['compliance_score']}/10")
|
||||
print(f" 风险项: {len(result['risks'])} 处")
|
||||
else:
|
||||
print(json.dumps(result, ensure_ascii=False, indent=2))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
@ -0,0 +1,6 @@
|
|||
# 子赛题四·科研辅助算法层依赖
|
||||
# 数据采集/报告/匹配/进度/合规/洞悉 仅用 Python 标准库(subprocess/json/sqlite3/re/datetime),无需第三方库。
|
||||
# 实测真正被 import 的第三方库只有以下两个:
|
||||
networkx>=3.1 # S2 科研知识图谱构建
|
||||
plotly>=5.18 # S6 科研成果可视化(交互 HTML)
|
||||
# (如后续 S4 升级为 TF-IDF、S6 改用模板渲染,再按需加 scikit-learn / jinja2)
|
||||
|
|
@ -0,0 +1,272 @@
|
|||
"""test_graph_build.py — S2 知识图谱的纯单元测试(不联网、不调 gitlink-cli)。
|
||||
|
||||
把「取数」与「算法」分离:build_graph() 只吃已构造好的 mock 数据结构。
|
||||
运行:`python test_graph_build.py` 或 `pytest scripts/research/`
|
||||
"""
|
||||
import os
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
import graph_build as G # noqa: E402
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# mock 数据构造
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _mock_repos():
|
||||
return [
|
||||
{
|
||||
"fullname": "alice/mindspore-vision",
|
||||
"identifier": "mindspore-vision",
|
||||
"author": {"login": "alice"},
|
||||
"description": "deep learning library for computer vision, "
|
||||
"image classification and object detection using CNN",
|
||||
"language": {"name": "Python"},
|
||||
"praises_count": 120,
|
||||
"forked_count": 30,
|
||||
},
|
||||
{
|
||||
"fullname": "bob/mindnlp",
|
||||
"identifier": "mindnlp",
|
||||
"author": {"login": "bob"},
|
||||
"description": "A nlp library: transformer, bert, text classification",
|
||||
"language": {"name": "Python"},
|
||||
"praises_count": 200,
|
||||
"forked_count": 50,
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
def _mock_contributors_map():
|
||||
# alice 既贡献自己的 repo,也贡献 bob 的 repo(→ 核心学者 + collaborates_with)
|
||||
return {
|
||||
"alice/mindspore-vision": [
|
||||
{"login": "alice", "contribution_perc": "60.0%"},
|
||||
{"login": "carol", "contribution_perc": "40.0%"},
|
||||
{"login": "i-robot", "contribution_perc": "0.0%"}, # bot,应被过滤
|
||||
],
|
||||
"bob/mindnlp": [
|
||||
{"login": "alice", "contribution_perc": "10.0%"},
|
||||
{"login": "bob", "contribution_perc": "90.0%"},
|
||||
{"login": "dependabot", "contribution_perc": "1.0%"}, # bot
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def _mock_readmes():
|
||||
return {
|
||||
"alice/mindspore-vision": "pytorch and resnet for object detection",
|
||||
"bob/mindnlp": "pretrain gpt and llm models",
|
||||
}
|
||||
|
||||
|
||||
def _build():
|
||||
return G.build_graph(_mock_repos(), _mock_contributors_map(),
|
||||
{}, _mock_readmes(), keywords=["vision", "nlp"])
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 工具函数
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_is_bot():
|
||||
assert G.is_bot("i-robot") is True
|
||||
assert G.is_bot("dependabot[bot]") is True
|
||||
assert G.is_bot("alice") is False
|
||||
assert G.is_bot("") is True
|
||||
|
||||
|
||||
def test_parse_ratio():
|
||||
assert G.parse_ratio("60.0%") == 0.6
|
||||
assert abs(G.parse_ratio("0.5") - 0.5) < 1e-9
|
||||
assert G.parse_ratio(0.3) == 0.3
|
||||
assert G.parse_ratio(None) == 0.0
|
||||
assert G.parse_ratio("n/a") == 0.0
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 建图:节点计数
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_node_counts():
|
||||
result = _build()
|
||||
by_type = {}
|
||||
for n in result["nodes"]:
|
||||
by_type[n["type"]] = by_type.get(n["type"], 0) + 1
|
||||
assert by_type.get("repo", 0) == 2
|
||||
# alice / carol / bob = 3 个学者(bot 被过滤)
|
||||
assert by_type.get("scholar", 0) == 3
|
||||
# 主题:CV + deep_learning + nlp = 至少 3 个
|
||||
assert by_type.get("topic", 0) >= 3
|
||||
assert result["meta"]["repo_count"] == 2
|
||||
assert result["meta"]["scholar_count"] == 3
|
||||
|
||||
|
||||
def test_repo_node_props():
|
||||
result = _build()
|
||||
repo_nodes = [n for n in result["nodes"] if n["type"] == "repo"]
|
||||
labels = {n["label"] for n in repo_nodes}
|
||||
assert "alice/mindspore-vision" in labels
|
||||
r = [n for n in repo_nodes if n["label"] == "alice/mindspore-vision"][0]
|
||||
assert r["props"]["language"] == "Python"
|
||||
assert r["props"]["stars"] == 120
|
||||
assert r["props"]["forks"] == 30
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 建图:边
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_contributes_to_weight():
|
||||
result = _build()
|
||||
edges = result["edges"]
|
||||
contrib = [e for e in edges if e["type"] == "contributes_to"
|
||||
and e["source"] == "scholar:alice"
|
||||
and e["target"] == "repo:alice/mindspore-vision"]
|
||||
assert contrib, "应有 alice→自己repo 的 contributes_to 边"
|
||||
# 60% → 0.6
|
||||
assert abs(contrib[0]["weight"] - 0.6) < 1e-9
|
||||
|
||||
|
||||
def test_owns_edge():
|
||||
result = _build()
|
||||
owns = [e for e in result["edges"] if e["type"] == "owns"
|
||||
and e["source"] == "scholar:bob"
|
||||
and e["target"] == "repo:bob/mindnlp"]
|
||||
assert owns, "bob 应有 owns 边到自己的 repo"
|
||||
|
||||
|
||||
def test_covers_topic_weight_range():
|
||||
"""covers_topic 权重应在 (0, 1] 且 <= 1。"""
|
||||
result = _build()
|
||||
covers = [e for e in result["edges"] if e["type"] == "covers_topic"]
|
||||
assert covers, "应至少有一条 covers_topic 边"
|
||||
for e in covers:
|
||||
assert 0.0 <= e["weight"] <= 1.0
|
||||
# 命中 computer_vision 主题(description 含 object detection/image classification)
|
||||
cv_edges = [e for e in covers if e["target"] == "topic:computer_vision"]
|
||||
assert cv_edges, "应识别出 computer_vision 主题"
|
||||
|
||||
|
||||
def test_collaborates_with():
|
||||
"""alice 与 carol 同在 alice/mindspore-vision → 至少一条 collaborates_with。"""
|
||||
result = _build()
|
||||
collab = [e for e in result["edges"] if e["type"] == "collaborates_with"]
|
||||
assert collab, "应有 collaborates_with 边"
|
||||
pair = {e["source"] for e in collab}
|
||||
assert "scholar:alice" in pair
|
||||
|
||||
|
||||
def test_related_to_when_topic_cooccur():
|
||||
"""computer_vision 与 deep_learning 在同一 repo 共现 → related_to。"""
|
||||
result = _build()
|
||||
related = [e for e in result["edges"] if e["type"] == "related_to"
|
||||
and e["source"] == "topic:computer_vision"]
|
||||
# mindspore-vision 的 description 同时命中 CV + deep_learning
|
||||
assert related, "共现主题应有 related_to 边"
|
||||
assert any(e["target"] == "topic:deep_learning" for e in related)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 衍生统计
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_core_scholars():
|
||||
result = _build()
|
||||
# alice 出现在 2 个 repo → 排首位
|
||||
top = result["core_scholars"][0]
|
||||
assert top["login"] == "alice"
|
||||
assert top["repo_count"] == 2
|
||||
|
||||
|
||||
def test_topic_heat_top():
|
||||
result = _build()
|
||||
heat = result["topic_heat"]
|
||||
assert heat, "应有主题热度榜"
|
||||
# deep_learning 在两个 repo 都命中 → 应在前列
|
||||
topics = [h["topic"] for h in heat]
|
||||
assert "deep_learning" in topics
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 渲染
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_render_mermaid_header():
|
||||
result = _build()
|
||||
mmd = G.render_mermaid(result)
|
||||
assert "graph TD" in mmd
|
||||
assert "classDef" in mmd # 着色定义
|
||||
assert "```" in mmd
|
||||
|
||||
|
||||
def test_render_mermaid_node_limit():
|
||||
"""节点过多时应被截断到 node_limit。"""
|
||||
big_repos = []
|
||||
big_contribs = {}
|
||||
big_readmes = {}
|
||||
for i in range(60):
|
||||
fn = f"u{i}/repo{i}"
|
||||
big_repos.append({
|
||||
"fullname": fn, "identifier": f"repo{i}",
|
||||
"author": {"login": f"u{i}"}, "description": "deep learning",
|
||||
"language": {"name": "Python"}, "praises_count": 0, "forked_count": 0,
|
||||
})
|
||||
big_contribs[fn] = [{"login": f"u{i}", "contribution_perc": "100%"}]
|
||||
big_readmes[fn] = "deep learning"
|
||||
result = G.build_graph(big_repos, big_contribs, {}, big_readmes,
|
||||
keywords=["dl"])
|
||||
mmd = G.render_mermaid(result, node_limit=40)
|
||||
# mermaid 里出现的节点声明数应 <= 40
|
||||
node_lines = [ln for ln in mmd.splitlines() if '["' in ln and "-->" not in ln]
|
||||
assert len(node_lines) <= 40
|
||||
|
||||
|
||||
def test_render_dot_header():
|
||||
result = _build()
|
||||
dot = G.render_dot(result)
|
||||
assert dot.startswith("digraph G")
|
||||
assert "fillcolor" in dot
|
||||
|
||||
|
||||
def test_render_report_sections():
|
||||
result = _build()
|
||||
md = G.render_report(result)
|
||||
assert "知识图谱报告" in md
|
||||
assert "主题热度榜" in md
|
||||
assert "核心学者" in md
|
||||
assert "deep_learning" in md or "computer_vision" in md
|
||||
|
||||
|
||||
def test_render_report_empty_safe():
|
||||
result = {"scenario": "S2", "keywords": [], "nodes": [], "edges": [],
|
||||
"core_scholars": [], "core_teams": [], "topic_heat": [],
|
||||
"meta": {"repo_count": 0, "node_count": 0, "edge_count": 0,
|
||||
"scholar_count": 0, "topic_count": 0}}
|
||||
md = G.render_report(result)
|
||||
assert "知识图谱报告" in md
|
||||
assert "未识别" in md or "暂无" in md
|
||||
|
||||
|
||||
def test_empty_inputs():
|
||||
"""空输入不应抛异常。"""
|
||||
result = G.build_graph([], {}, {}, {})
|
||||
assert result["meta"]["node_count"] == 0
|
||||
assert result["meta"]["edge_count"] == 0
|
||||
assert result["nodes"] == []
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _run_all():
|
||||
fns = [v for k, v in sorted(globals().items()) if k.startswith("test_")]
|
||||
for fn in fns:
|
||||
fn()
|
||||
print(f"PASS {fn.__name__}")
|
||||
print(f"\nAll {len(fns)} graph_build tests passed.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
_run_all()
|
||||
|
|
@ -0,0 +1,80 @@
|
|||
"""test_helpers.py — gitlink_data / collect 归一化工具的纯单元测试(不联网)。
|
||||
|
||||
运行:`pytest scripts/research/` 或 `python scripts/research/test_helpers.py`
|
||||
"""
|
||||
import sys
|
||||
import os
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
import gitlink_data as gd # noqa: E402
|
||||
import collect # noqa: E402
|
||||
|
||||
|
||||
def test_first_list_list_input():
|
||||
assert gd.first_list([1, 2, 3]) == [1, 2, 3]
|
||||
|
||||
|
||||
def test_first_list_known_key():
|
||||
assert gd.first_list({"projects": [{"id": 1}], "total_count": 1}) == [{"id": 1}]
|
||||
|
||||
|
||||
def test_first_list_single_list_value():
|
||||
# 无已知键但只有一个列表值时,兜底返回它
|
||||
assert gd.first_list({"whatever": [9, 9]}) == [9, 9]
|
||||
|
||||
|
||||
def test_first_list_empty():
|
||||
assert gd.first_list({"total_count": 0}) == []
|
||||
assert gd.first_list(None) == []
|
||||
|
||||
|
||||
def test_total_count():
|
||||
assert gd.total_count({"total_count": 42}) == 42
|
||||
assert gd.total_count({"totalCount": 7}) == 7
|
||||
assert gd.total_count({"projects": []}) is None
|
||||
|
||||
|
||||
def test_login_of_flat():
|
||||
assert collect.login_of({"login": "whale"}) == "whale"
|
||||
|
||||
|
||||
def test_login_of_nested_author():
|
||||
assert collect.login_of({"author": {"login": "baoerjun"}}) == "baoerjun"
|
||||
|
||||
|
||||
def test_login_of_name_fallback():
|
||||
assert collect.login_of({"name": "surponess"}) == "surponess"
|
||||
|
||||
|
||||
def test_login_of_empty():
|
||||
assert collect.login_of({}) == ""
|
||||
assert collect.login_of("not a dict") == ""
|
||||
|
||||
|
||||
def test_repo_fullname_with_author():
|
||||
project = {"identifier": "gitlink-cli", "author": {"login": "whale_hihihi"}}
|
||||
assert collect.repo_fullname(project) == "whale_hihihi/gitlink-cli"
|
||||
|
||||
|
||||
def test_repo_fullname_missing_owner():
|
||||
assert collect.repo_fullname({"identifier": "foo"}) == "foo"
|
||||
|
||||
|
||||
def test_as_int_as_float():
|
||||
assert collect.as_int("12") == 12
|
||||
assert collect.as_int(None) == 0
|
||||
assert collect.as_float("1.5") == 1.5
|
||||
assert collect.as_float("x", -1.0) == -1.0
|
||||
|
||||
|
||||
def _run_all():
|
||||
fns = [v for k, v in sorted(globals().items()) if k.startswith("test_")]
|
||||
for fn in fns:
|
||||
fn()
|
||||
print(f"PASS {fn.__name__}")
|
||||
print(f"\nAll {len(fns)} helper tests passed.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
_run_all()
|
||||
|
|
@ -0,0 +1,260 @@
|
|||
"""test_lineage.py — lineage.py 纯函数单元测试。
|
||||
|
||||
不联网、不调 gitlink-cli。把「取数」与「算法」分离:算法函数接收已取好的
|
||||
Python 数据结构,本测试用 mock 数据喂算法。
|
||||
|
||||
运行: python test_lineage.py
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
import unittest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
# 仅 import 算法纯函数,绝不触发联网取数
|
||||
import lineage as L # noqa: E402
|
||||
|
||||
|
||||
class TestExperimentClassifier(unittest.TestCase):
|
||||
def test_is_experiment_file_hits(self):
|
||||
cases = [
|
||||
"experiments/mnist/run.py",
|
||||
"experiment_train.py",
|
||||
"benchmark/imagenet/eval.py",
|
||||
"eval/metrics.py",
|
||||
"tests/test_model.py",
|
||||
"test/foo.py",
|
||||
"data/dataset.csv",
|
||||
"src/benchmark_infer.py",
|
||||
]
|
||||
for p in cases:
|
||||
with self.subTest(p=p):
|
||||
self.assertTrue(L.is_experiment_file(p), f"应判为实验文件: {p}")
|
||||
|
||||
def test_is_experiment_file_misses(self):
|
||||
cases = [
|
||||
"src/model.py",
|
||||
"README.md",
|
||||
"docs/index.md",
|
||||
"main.go",
|
||||
"config.yaml",
|
||||
"pkg/utils/util.py",
|
||||
]
|
||||
for p in cases:
|
||||
with self.subTest(p=p):
|
||||
self.assertFalse(L.is_experiment_file(p), f"不应判为实验文件: {p}")
|
||||
|
||||
def test_empty(self):
|
||||
self.assertFalse(L.is_experiment_file(""))
|
||||
self.assertFalse(L.is_experiment_file(None)) # type: ignore[arg-type]
|
||||
|
||||
|
||||
class TestDocClassifier(unittest.TestCase):
|
||||
def test_md_anywhere(self):
|
||||
self.assertTrue(L.is_doc_file("README.md"))
|
||||
self.assertTrue(L.is_doc_file("docs/guide.md"))
|
||||
self.assertTrue(L.is_doc_file("deep/nested/notes.md"))
|
||||
self.assertTrue(L.is_doc_file("GUIDE.MD")) # 大小写不敏感
|
||||
|
||||
def test_docs_dir(self):
|
||||
self.assertTrue(L.is_doc_file("docs/index.html"))
|
||||
self.assertTrue(L.is_doc_file("docs/config.yaml"))
|
||||
self.assertTrue(L.is_doc_file("a/docs/b.txt"))
|
||||
|
||||
def test_non_doc(self):
|
||||
self.assertFalse(L.is_doc_file("src/main.py"))
|
||||
self.assertFalse(L.is_doc_file("tests/x.py"))
|
||||
self.assertFalse(L.is_doc_file("data.csv"))
|
||||
self.assertFalse(L.is_doc_file(""))
|
||||
|
||||
|
||||
class TestTimestampParse(unittest.TestCase):
|
||||
def test_iso_string(self):
|
||||
t = L._to_epoch("2024-05-01T08:00:00Z")
|
||||
self.assertGreater(t, 0)
|
||||
# 反解回日期
|
||||
self.assertTrue(L._iso_date(t).startswith("2024-05-01"))
|
||||
|
||||
def test_int_seconds(self):
|
||||
self.assertAlmostEqual(L._to_epoch(1714521600), 1714521600.0)
|
||||
|
||||
def test_int_millis(self):
|
||||
self.assertAlmostEqual(L._to_epoch(1714521600000), 1714521600.0)
|
||||
|
||||
def test_garbage(self):
|
||||
self.assertEqual(L._to_epoch(None), 0.0)
|
||||
self.assertEqual(L._to_epoch(""), 0.0)
|
||||
self.assertEqual(L._to_epoch("not-a-date"), 0.0)
|
||||
|
||||
def test_iso_date_zero(self):
|
||||
self.assertEqual(L._iso_date(0.0), "")
|
||||
|
||||
|
||||
class TestBranchMap(unittest.TestCase):
|
||||
def test_single_default_branch(self):
|
||||
commits = [
|
||||
{"timestamp": "2024-01-01T00:00:00Z"},
|
||||
{"timestamp": "2024-06-01T00:00:00Z"},
|
||||
{"timestamp": "2024-03-01T00:00:00Z"},
|
||||
]
|
||||
bm = L.build_branch_map(commits, "master")
|
||||
self.assertEqual(len(bm), 1)
|
||||
b = bm[0]
|
||||
self.assertEqual(b["name"], "master")
|
||||
self.assertTrue(b["is_default"])
|
||||
self.assertEqual(b["commits"], 3)
|
||||
# 最后活跃应取最大值 2024-06-01
|
||||
self.assertTrue(b["last_active"].startswith("2024-06-01"))
|
||||
|
||||
def test_empty(self):
|
||||
bm = L.build_branch_map([], "main")
|
||||
self.assertEqual(bm, [{"name": "main", "commits": 0,
|
||||
"last_active": "", "is_default": True}])
|
||||
|
||||
def test_non_list(self):
|
||||
bm = L.build_branch_map(None, "main") # type: ignore[arg-type]
|
||||
self.assertEqual(bm[0]["commits"], 0)
|
||||
|
||||
|
||||
class TestCommitTimeline(unittest.TestCase):
|
||||
def test_aggregation_and_sort(self):
|
||||
commits = [
|
||||
{"timestamp": "2024-01-02T00:00:00Z"},
|
||||
{"timestamp": "2024-01-02T12:00:00Z"},
|
||||
{"timestamp": "2024-01-01T00:00:00Z"},
|
||||
]
|
||||
tl = L.commit_timeline(commits, bucket="day")
|
||||
self.assertEqual(tl, [
|
||||
{"date": "2024-01-01", "count": 1},
|
||||
{"date": "2024-01-02", "count": 2},
|
||||
])
|
||||
|
||||
def test_skips_garbage(self):
|
||||
commits = [{"timestamp": "bad"}, {"timestamp": ""}]
|
||||
self.assertEqual(L.commit_timeline(commits), [])
|
||||
|
||||
|
||||
class TestPrMergePatterns(unittest.TestCase):
|
||||
def test_extract_and_sort(self):
|
||||
prs = [
|
||||
{"index": 10, "title": "feat A", "status": 1,
|
||||
"pr_created_unix": 1717200000, "changed_files": 5},
|
||||
{"index": 2, "title": "feat B", "status": 1,
|
||||
"pr_merged_unix": 1714521600, "changed_files": 20},
|
||||
{"index": 5, "title": "feat C", "status": 1,
|
||||
"pr_created_unix": 1715000000, "additions": 3, "deletions": 4},
|
||||
]
|
||||
out = L.pr_merge_patterns(prs)
|
||||
self.assertEqual(len(out), 3)
|
||||
# 升序:1714521600(2024-05-01) < 1715000000 < 1717200000
|
||||
self.assertEqual(out[0]["number"], 2)
|
||||
self.assertEqual(out[0]["changed_files"], 20)
|
||||
self.assertTrue(out[0]["merged_time"].startswith("2024-05"))
|
||||
# additions/deletions 兜底近似
|
||||
last = next(p for p in out if p["number"] == 5)
|
||||
self.assertEqual(last["changed_files"], 7)
|
||||
# 空 merged_time 排末尾
|
||||
no_time = L.pr_merge_patterns([
|
||||
{"index": 1, "title": "x", "status": 1},
|
||||
{"index": 2, "title": "y", "status": 1, "pr_created_unix": 1714521600},
|
||||
])
|
||||
self.assertEqual(no_time[-1]["number"], 1)
|
||||
|
||||
def test_empty(self):
|
||||
self.assertEqual(L.pr_merge_patterns([]), [])
|
||||
|
||||
|
||||
class TestDocEvolution(unittest.TestCase):
|
||||
def test_filter_docs(self):
|
||||
tree = [
|
||||
{"name": "guide.md", "path": "docs/guide.md", "date": "2024-03-01"},
|
||||
{"name": "index.html", "path": "docs/index.html"},
|
||||
{"name": "model.py", "path": "src/model.py"}, # 排除
|
||||
{"name": "old_2022-01-01.md", "path": "docs/old_2022-01-01.md"},
|
||||
]
|
||||
out = L.doc_evolution(tree)
|
||||
files = {d["file"] for d in out}
|
||||
self.assertIn("guide.md", files)
|
||||
self.assertIn("index.html", files)
|
||||
self.assertIn("old_2022-01-01.md", files)
|
||||
self.assertNotIn("model.py", files)
|
||||
# 文件名日期回退
|
||||
old = next(d for d in out if d["file"] == "old_2022-01-01.md")
|
||||
self.assertEqual(old["last_date"], "2022-01-01")
|
||||
# 显式 date 字段优先
|
||||
g = next(d for d in out if d["file"] == "guide.md")
|
||||
self.assertEqual(g["last_date"], "2024-03-01")
|
||||
|
||||
|
||||
class TestInnovationPoints(unittest.TestCase):
|
||||
def test_high_impact_and_milestone(self):
|
||||
prs = [
|
||||
{"index": 1, "title": "chore: typo", "status": 1,
|
||||
"pr_created_unix": 1714521600, "changed_files": 2}, # 低影响,无关键词
|
||||
{"index": 2, "title": "feat: add transformer model", "status": 1,
|
||||
"pr_created_unix": 1715000000, "changed_files": 25}, # 大规模+关键词
|
||||
{"index": 3, "title": "implement benchmark suite", "status": 1,
|
||||
"pr_created_unix": 1717200000, "changed_files": 4}, # 仅关键词
|
||||
]
|
||||
out = L.innovation_points(prs, commits=[])
|
||||
# 应识别出 PR#2 和 PR#3,PR#1 被过滤
|
||||
nums = sorted(it["description"] for it in out)
|
||||
self.assertTrue(any("transformer" in n.lower() for n in nums))
|
||||
self.assertTrue(any("benchmark" in n.lower() for n in nums))
|
||||
cats = [it["category"] for it in out]
|
||||
self.assertIn("大规模重构/新特性", cats)
|
||||
self.assertIn("特性引入", cats)
|
||||
# 大规模 PR 排前(impact 更高)
|
||||
self.assertIn("transformer", out[0]["description"].lower())
|
||||
# 每条都带证据
|
||||
for it in out:
|
||||
self.assertTrue(it["evidence"])
|
||||
self.assertTrue(it["category"])
|
||||
|
||||
def test_empty(self):
|
||||
self.assertEqual(L.innovation_points([], []), [])
|
||||
|
||||
def test_top_limit(self):
|
||||
prs = [{"index": i, "title": f"add feature {i}", "status": 1,
|
||||
"pr_created_unix": 1714521600 + i * 86400, "changed_files": 15}
|
||||
for i in range(20)]
|
||||
out = L.innovation_points(prs, commits=[], top=5)
|
||||
self.assertEqual(len(out), 5)
|
||||
|
||||
|
||||
class TestRender(unittest.TestCase):
|
||||
"""渲染函数不抛异常、产出非空。"""
|
||||
|
||||
def _result(self):
|
||||
return {
|
||||
"scenario": "S1_repository_research_insight",
|
||||
"repo": "o/r", "default_branch": "master",
|
||||
"commit_timeline": [{"date": "2024-01-01", "count": 3}],
|
||||
"branch_map": [{"name": "master", "commits": 5, "last_active": "2024-06-01",
|
||||
"is_default": True}],
|
||||
"pr_merge_patterns": [{"number": 2, "title": "feat A", "status": 1,
|
||||
"merged_time": "2024-06-01", "changed_files": 9}],
|
||||
"doc_evolution": [{"file": "guide.md", "last_date": "2024-03-01"}],
|
||||
"experiment_files": ["benchmark/eval.py"],
|
||||
"innovation_points": [{"description": "feat A", "evidence": "PR #2",
|
||||
"category": "特性引入"}],
|
||||
"meta": {"commit_count": 5, "merged_pr_count": 1, "doc_count": 1,
|
||||
"experiment_file_count": 1},
|
||||
}
|
||||
|
||||
def test_report(self):
|
||||
r = L.render_report(self._result())
|
||||
self.assertIn("仓库级科研项目洞悉报告", r)
|
||||
self.assertIn("master", r)
|
||||
self.assertIn("feat A", r)
|
||||
|
||||
def test_mermaid(self):
|
||||
m = L.render_mermaid(self._result())
|
||||
self.assertIn("gitGraph", m)
|
||||
self.assertIn("master", m)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main(verbosity=2)
|
||||
|
|
@ -0,0 +1,77 @@
|
|||
"""test_match.py — S4 协作匹配的纯单元测试(不联网)。
|
||||
|
||||
运行:`pytest scripts/research/` 或 `python scripts/research/test_match.py`
|
||||
"""
|
||||
import os
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
import match as M # noqa: E402
|
||||
|
||||
|
||||
def test_cosine_basic():
|
||||
assert M.cosine({"a": 1, "b": 0}, {"a": 1, "b": 0}) == 1.0
|
||||
assert M.cosine({"a": 1}, {"b": 1}) == 0.0
|
||||
# 对称且在 [0,1]
|
||||
v = M.cosine({"a": 1, "b": 2}, {"a": 2, "b": 1})
|
||||
assert 0.0 < v < 1.0
|
||||
|
||||
|
||||
def test_cosine_empty_safe():
|
||||
assert M.cosine({}, {"a": 1}) == 0.0
|
||||
assert M.cosine({}, {}) == 0.0
|
||||
|
||||
|
||||
def test_jaccard():
|
||||
assert M.jaccard(["python", "go"], ["python", "rust"]) == 1 / 3
|
||||
assert M.jaccard([], ["x"]) == 0.0
|
||||
|
||||
|
||||
def test_priority_weight():
|
||||
assert M._priority_weight({"priority_name": "高"}) == 3.0
|
||||
assert M._priority_weight({"priority_name": "urgent"}) == 3.0
|
||||
assert M._priority_weight({"priority_name": "普通"}) == 2.0
|
||||
assert M._priority_weight({"priority_name": "低"}) == 1.0
|
||||
assert M._priority_weight({}) == 1.0
|
||||
|
||||
|
||||
def test_parse_ratio_percent_string():
|
||||
assert M._parse_ratio("1.18%") == 0.0118
|
||||
assert abs(M._parse_ratio("50%") - 0.5) < 1e-9
|
||||
|
||||
|
||||
def test_parse_ratio_plain():
|
||||
assert M._parse_ratio(0.5) == 0.5
|
||||
assert M._parse_ratio("0.2") == 0.2
|
||||
assert M._parse_ratio(None) == 0.0
|
||||
assert M._parse_ratio("n/a") == 0.0
|
||||
|
||||
|
||||
def test_render_report_has_sections():
|
||||
result = {
|
||||
"repo": "o/r", "gap_topics": ["deep_learning"], "needed_languages": ["python"],
|
||||
"gap_signals": [{"type": "unresolved_issue", "topic": "deep_learning",
|
||||
"evidence": "x", "priority": "高"}],
|
||||
"candidates": [{"login": "alice", "score": 20.0, "topic_overlap": 0.5,
|
||||
"language_match": 0.5, "activity_level": "high",
|
||||
"repo_languages": ["python"], "reasons": ["覆盖缺口主题"]}],
|
||||
"meta": {"pool_size": 1, "issue_sample": 1},
|
||||
}
|
||||
md = M.render_report(result)
|
||||
assert "缺口分析" in md or "技术缺口" in md
|
||||
assert "alice" in md
|
||||
mm = M.render_mermaid(result)
|
||||
assert mm.startswith("```mermaid") and "alice" in mm
|
||||
|
||||
|
||||
def _run_all():
|
||||
fns = [v for k, v in sorted(globals().items()) if k.startswith("test_")]
|
||||
for fn in fns:
|
||||
fn()
|
||||
print(f"PASS {fn.__name__}")
|
||||
print(f"\nAll {len(fns)} match tests passed.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
_run_all()
|
||||
|
|
@ -0,0 +1,150 @@
|
|||
"""test_report.py — S5 进度跟踪与预警的纯单元测试(不联网)。
|
||||
|
||||
运行:`pytest scripts/research/` 或 `python scripts/research/test_report.py`
|
||||
"""
|
||||
import os
|
||||
import sys
|
||||
from datetime import datetime, timedelta, timezone
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
import report as R # noqa: E402
|
||||
|
||||
UTC = timezone.utc
|
||||
NOW = datetime(2026, 6, 29, 12, 0, 0, tzinfo=UTC)
|
||||
|
||||
|
||||
def iso(dt):
|
||||
return dt.isoformat()
|
||||
|
||||
|
||||
def test_parse_time_iso():
|
||||
assert R.parse_time("2026-06-01T08:30:00+00:00").year == 2026
|
||||
assert R.parse_time("2026-06-01T08:30:00Z").month == 6
|
||||
|
||||
|
||||
def test_parse_time_unix():
|
||||
# 整数与整数字串都按秒级时间戳解析
|
||||
dt = R.parse_time(1717200000)
|
||||
assert dt is not None and dt.year == 2024
|
||||
assert R.parse_time("1717200000").year == 2024
|
||||
|
||||
|
||||
def test_parse_time_none_and_garbage():
|
||||
assert R.parse_time(None) is None
|
||||
assert R.parse_time("") is None
|
||||
assert R.parse_time("not a date") is None
|
||||
|
||||
|
||||
def test_in_window():
|
||||
lo = NOW - timedelta(days=7)
|
||||
assert R.in_window(NOW - timedelta(days=2), 7, NOW) is True
|
||||
assert R.in_window(NOW - timedelta(days=10), 7, NOW) is False
|
||||
assert R.in_window(None, 7, NOW) is False
|
||||
|
||||
|
||||
def test_week_stats_splits_this_and_last_week():
|
||||
commits = [
|
||||
{"timestamp": iso(NOW - timedelta(days=2)), "author": {"login": "a"}}, # 本周
|
||||
{"timestamp": iso(NOW - timedelta(days=3)), "author": {"login": "b"}}, # 本周
|
||||
{"timestamp": iso(NOW - timedelta(days=10)), "author": {"login": "a"}}, # 上周
|
||||
]
|
||||
issues = [
|
||||
{"created_at": iso(NOW - timedelta(days=1)), "status": "open"}, # 本周新增
|
||||
{"created_at": iso(NOW - timedelta(days=9)), "status": "open"}, # 上周新增
|
||||
]
|
||||
prs = []
|
||||
stats = R.week_stats(commits, issues, prs, ["a", "b"], NOW)
|
||||
assert stats["this_week"]["commits"] == 2
|
||||
assert stats["last_week"]["commits"] == 1
|
||||
assert stats["this_week"]["issues_opened"] == 1
|
||||
assert stats["last_week"]["issues_opened"] == 1
|
||||
|
||||
|
||||
def test_week_stats_stale_issue():
|
||||
# 开放、最近活动 > 30 天 → stale
|
||||
old = NOW - timedelta(days=40)
|
||||
issues = [{"status": "open", "created_at": iso(old), "journals_updated_at": iso(old)}]
|
||||
stats = R.week_stats([], issues, [], [], NOW)
|
||||
assert stats["this_week"]["issues_stale"] == 1
|
||||
|
||||
|
||||
def test_trend():
|
||||
assert R.trend({"commits": 10}, {"commits": 5})["activity_level"] == "increasing"
|
||||
assert R.trend({"commits": 2}, {"commits": 10})["activity_level"] == "decreasing"
|
||||
assert R.trend({"commits": 10}, {"commits": 10})["activity_level"] == "stable"
|
||||
# 上周为 0、本周有提交 → 100%
|
||||
assert R.trend({"commits": 3}, {"commits": 0})["commit_delta_pct"] == 100.0
|
||||
|
||||
|
||||
def test_risk_low_activity():
|
||||
stats = {"this_week": {"commits": 1, "issues_stale": 0, "prs_open_stale": 0}}
|
||||
warns = R.risk_warnings(stats, [], [], commits=None, now=NOW)
|
||||
assert any(w["type"] == "low_activity" for w in warns)
|
||||
|
||||
|
||||
def test_risk_bus_factor():
|
||||
# 一人占本周全部提交 → bus factor
|
||||
commits = [{"timestamp": iso(NOW - timedelta(days=1)), "author": {"login": "only"}} for _ in range(5)]
|
||||
stats = {"this_week": {"commits": 5, "issues_stale": 0, "prs_open_stale": 0}}
|
||||
warns = R.risk_warnings(stats, [], [], commits=commits, now=NOW)
|
||||
assert any(w["type"] == "bus_factor" for w in warns)
|
||||
|
||||
|
||||
def test_milestone_progress_overdue():
|
||||
ms = [{"name": "v1.0", "status": "open", "effective_date": iso(NOW - timedelta(days=5))}]
|
||||
issues = [
|
||||
{"milestone_name": "v1.0", "status": "closed"},
|
||||
{"milestone_name": "v1.0", "status": "open"},
|
||||
{"milestone_name": "v1.0", "status": "open"},
|
||||
]
|
||||
out = R.milestone_progress(ms, issues, NOW)
|
||||
assert len(out) == 1
|
||||
assert out[0]["total"] == 3 and out[0]["closed"] == 1
|
||||
assert out[0]["completion_pct"] == round(100 / 3, 1)
|
||||
assert out[0]["overdue"] is True
|
||||
|
||||
|
||||
def test_render_report_contains_sections():
|
||||
# 直接构造一个最小 result 喂渲染器(不触网)
|
||||
res = {
|
||||
"repo": "o/r", "generated_at": iso(NOW),
|
||||
"week_stats": {"this_week": {"commits": 1, "issues_opened": 0, "issues_closed": 0,
|
||||
"issues_stale": 0, "prs_opened": 0, "prs_merged": 0,
|
||||
"prs_open_stale": 0, "contributors_active": 1},
|
||||
"last_week": {"commits": 0, "issues_opened": 0, "issues_closed": 0,
|
||||
"issues_stale": 0, "prs_opened": 0, "prs_merged": 0,
|
||||
"prs_open_stale": 0, "contributors_active": 0},
|
||||
"window": {"this_week_start": iso(NOW), "now": iso(NOW),
|
||||
"last_week_start": iso(NOW), "last_week_end": iso(NOW)},
|
||||
"total_contributors": 1},
|
||||
"trend": {"commit_delta_pct": 100.0, "activity_level": "increasing"},
|
||||
"milestones": [], "risk_warnings": [],
|
||||
"meta": {"commits_fetched": 1, "issues_fetched": 0, "prs_fetched": 0,
|
||||
"milestones_fetched": 0, "contributors_fetched": 1},
|
||||
}
|
||||
md = R.render_report(res)
|
||||
assert "周报" in md and "趋势" in md
|
||||
|
||||
|
||||
def test_pr_status_string_and_int():
|
||||
# GitLink PR 列表 status 是字符串 'merged'/'open'/'closed'
|
||||
assert R._pr_status({"status": "merged"}) == 1
|
||||
assert R._pr_status({"status": "open"}) == 0
|
||||
assert R._pr_status({"status": "closed"}) == 2
|
||||
# 详情/health 可能给 pull_request_status 整数 0/1/2
|
||||
assert R._pr_status({"pull_request_status": 1}) == 1
|
||||
assert R._pr_status({"pull_request_status": 0}) == 0
|
||||
assert R._pr_status({"pull_request_status": 2}) == 2
|
||||
|
||||
|
||||
def _run_all():
|
||||
fns = [v for k, v in sorted(globals().items()) if k.startswith("test_")]
|
||||
for fn in fns:
|
||||
fn()
|
||||
print(f"PASS {fn.__name__}")
|
||||
print(f"\nAll {len(fns)} report tests passed.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
_run_all()
|
||||
|
|
@ -0,0 +1,278 @@
|
|||
"""test_repro.py — repro.py 的纯单元测试。
|
||||
|
||||
不联网、不调 gitlink-cli。把"取数"和"算法"分离:直接给算法函数喂 mock 数据。
|
||||
|
||||
运行: python test_repro.py
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
import repro # noqa: E402
|
||||
|
||||
# 用一个轻量测试框架,避免依赖 unittest(保持无第三方依赖)。
|
||||
_failures: list[str] = []
|
||||
|
||||
|
||||
def check(cond: bool, msg: str) -> None:
|
||||
status = "PASS" if cond else "FAIL"
|
||||
print(f" [{status}] {msg}")
|
||||
if not cond:
|
||||
_failures.append(msg)
|
||||
|
||||
|
||||
def check_eq(actual, expected, msg: str) -> None:
|
||||
ok = actual == expected
|
||||
status = "PASS" if ok else "FAIL"
|
||||
print(f" [{status}] {msg} (got={actual!r})")
|
||||
if not ok:
|
||||
_failures.append(f"{msg}: expected {expected!r}, got {actual!r}")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 1) identify_license
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_identify_license():
|
||||
print("\n== test_identify_license ==")
|
||||
cases = {
|
||||
"Apache License\nVersion 2.0": "Apache-2.0",
|
||||
"MIT License\n\nCopyright (c) 2024": "MIT",
|
||||
"GNU GENERAL PUBLIC LICENSE\nVersion 3": "GPL",
|
||||
"木兰宽松许可证, 第2版": "MulanPSL-2.0",
|
||||
"MulanPSL v2": "MulanPSL-2.0",
|
||||
"BSD 3-Clause License": "BSD",
|
||||
"ISC License": "ISC",
|
||||
"Mozilla Public License Version 2.0": "MPL",
|
||||
"random text without any license keyword": "None",
|
||||
"": "None",
|
||||
" \n ": "None",
|
||||
}
|
||||
for text, expected in cases.items():
|
||||
info = repro.identify_license(text)
|
||||
check_eq(info["license"], expected, f"identify_license({text[:24]!r})")
|
||||
# 识别标志
|
||||
check(repro.identify_license("MIT License")["recognized"] is True, "MIT recognized=True")
|
||||
check(repro.identify_license("nope")["recognized"] is False, "unknown recognized=False")
|
||||
# GPL 优先于 LGPL:LGPL 文本应命中 LGPL(因为 LGPL 模式在 GPL 之前)
|
||||
l = repro.identify_license("GNU Lesser General Public License v3")
|
||||
check_eq(l["license"], "LGPL", "LGPL not misidentified as GPL")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 2) scan_secrets
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_scan_secrets():
|
||||
print("\n== test_scan_secrets ==")
|
||||
sample = """\
|
||||
API_KEY=sk_live_abcdef1234567890abcd
|
||||
AWS_KEY=AKIAIOSFODNN7EXAMPLE
|
||||
mail: someone@example.com
|
||||
phone: 13812345678
|
||||
-----BEGIN RSA PRIVATE KEY-----
|
||||
MIIEpAIBAAKCAQEA...
|
||||
"""
|
||||
findings = repro.scan_secrets(sample, file="config.env")
|
||||
cats = {f["category"] for f in findings}
|
||||
check("generic_api_key" in cats, "detect generic_api_key")
|
||||
check("aws_access_key" in cats, "detect aws_access_key")
|
||||
check("email" in cats, "detect email")
|
||||
check("phone_cn" in cats, "detect phone_cn")
|
||||
check("private_key" in cats, "detect private_key")
|
||||
# 每条都有必要字段
|
||||
for f in findings:
|
||||
check(set(["level", "category", "file", "line", "detail"]).issubset(f.keys()),
|
||||
f"finding {f['category']} has required keys")
|
||||
check_eq(f["file"], "config.env", f"{f['category']} file field")
|
||||
# critical 级别存在
|
||||
levels = {f["level"] for f in findings}
|
||||
check("critical" in levels, "critical level present (private key/aws)")
|
||||
# 空文本
|
||||
check_eq(repro.scan_secrets(""), [], "empty text -> no findings")
|
||||
# 无敏感信息文本
|
||||
check_eq(repro.scan_secrets("hello world\nnothing here"), [], "clean text -> no findings")
|
||||
# 长串脱敏(detail 不超长)
|
||||
long_token = "api_key=" + "a" * 60
|
||||
f2 = repro.scan_secrets(long_token, file="x")
|
||||
if f2:
|
||||
check(len(f2[0]["detail"]) <= 50, "long secret is masked in detail")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 3) repro_checks
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_repro_checks():
|
||||
print("\n== test_repro_checks ==")
|
||||
file_texts = {
|
||||
"README.md": "# Project\n\n## Install\n pip install -e .\n## Build\ndocker build .\n"
|
||||
"Dataset: download from xxx. reproduce: python run.py\nUsage: see docs.",
|
||||
}
|
||||
tree = [
|
||||
{"path": ".gitea/workflows/ci.yml"},
|
||||
{"path": "go.sum"},
|
||||
{"path": "requirements.txt"},
|
||||
{"path": "Dockerfile"},
|
||||
{"path": "src/main.go"},
|
||||
]
|
||||
items = repro.repro_checks(file_texts, tree, repo_info={"default_branch": "master"})
|
||||
names = {it["name"]: it for it in items}
|
||||
check(names["CI 配置"]["pass"] is True, "CI detected")
|
||||
check_eq(names["CI 配置"]["score"], 2, "CI score=2")
|
||||
check(names["依赖锁文件"]["pass"] is True, "lockfile detected")
|
||||
check(names["README 复现说明"]["pass"] is True, "README repro keywords detected")
|
||||
check(names["README 复现说明"]["score"] >= 1, "README repro score>=1")
|
||||
check(names["容器化环境"]["pass"] is True, "Dockerfile detected")
|
||||
|
||||
# 无 CI / 无 lockfile 场景
|
||||
items2 = repro.repro_checks({}, [], repo_info={})
|
||||
names2 = {it["name"]: it for it in items2}
|
||||
check(names2["CI 配置"]["pass"] is False, "no CI -> fail")
|
||||
check_eq(names2["CI 配置"]["score"], 0, "no CI score=0")
|
||||
check(names2["依赖锁文件"]["pass"] is False, "no lockfile -> fail")
|
||||
check(names2["README 复现说明"]["pass"] is False, "no README -> fail")
|
||||
# 版本 tag:无 tag 字段 → score=1(不扣满分但提示)
|
||||
check_eq(names2["版本 tag"]["score"], 1, "unknown tag -> score=1")
|
||||
# 有 tag 字段 → score=2
|
||||
items3 = repro.repro_checks({}, [], repo_info={"version": "v1.2.3"})
|
||||
names3 = {it["name"]: it for it in items3}
|
||||
check_eq(names3["版本 tag"]["score"], 2, "version present -> score=2")
|
||||
check(names3["版本 tag"]["pass"] is True, "version present -> pass")
|
||||
|
||||
# score 范围合法
|
||||
for it in items:
|
||||
check(0 <= it["score"] <= 2, f"{it['name']} score in [0,2]")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 4) compliance_items
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_compliance_items():
|
||||
print("\n== test_compliance_items ==")
|
||||
license_info = {"license": "MIT", "recognized": True, "evidence": "MIT License"}
|
||||
file_texts = {"LICENSE": "MIT License\nCopyright (c) 2024 Test", "README.md": "see LICENSE"}
|
||||
tree = [
|
||||
{"path": "SECURITY.md"},
|
||||
{"path": "CONTRIBUTING.md"},
|
||||
{"path": "requirements.txt"},
|
||||
]
|
||||
items = repro.compliance_items(license_info, file_texts, tree)
|
||||
names = {it["name"]: it for it in items}
|
||||
check(names["LICENSE 文件"]["pass"] is True, "LICENSE recognized")
|
||||
check_eq(names["LICENSE 文件"]["score"], 2, "LICENSE score=2")
|
||||
check(names["安全策略 SECURITY.md"]["pass"] is True, "SECURITY.md present")
|
||||
check(names["版权声明"]["pass"] is True, "copyright present")
|
||||
check(names["贡献指南"]["pass"] is True, "CONTRIBUTING present")
|
||||
check(names["依赖清单声明"]["pass"] is True, "dep manifest present")
|
||||
|
||||
# 缺失场景
|
||||
license_none = {"license": "None", "recognized": False, "evidence": "missing"}
|
||||
items2 = repro.compliance_items(license_none, {}, [])
|
||||
names2 = {it["name"]: it for it in items2}
|
||||
check(names2["LICENSE 文件"]["pass"] is False, "no LICENSE -> fail")
|
||||
check_eq(names2["LICENSE 文件"]["score"], 0, "no LICENSE score=0")
|
||||
check(names2["安全策略 SECURITY.md"]["pass"] is False, "no SECURITY -> fail")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 5) data_privacy
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_data_privacy():
|
||||
print("\n== test_data_privacy ==")
|
||||
# 健康场景:无 data/,无 .env,.gitignore 忽略 .env
|
||||
tree = [{"path": "src/main.py"}, {"path": ".gitignore"}]
|
||||
dp = repro.data_privacy(tree, ".env\n*.key\nnode_modules/")
|
||||
items = {it["name"]: it for it in dp["items"]}
|
||||
check(items["数据目录入库"]["pass"] is True, "no data dir -> pass")
|
||||
check(items[".env 入库"]["pass"] is True, "no .env -> pass")
|
||||
check(items[".gitignore 忽略 .env"]["pass"] is True, "gitignore ignores .env")
|
||||
check_eq(len(dp["risks"]), 0, "healthy repo -> 0 risks")
|
||||
|
||||
# 风险场景:data/ 入库,.env 入库,gitignore 未忽略 .env
|
||||
tree2 = [{"path": "data/raw.csv"}, {"path": ".env"}, {"path": "config/secrets.yml"}]
|
||||
dp2 = repro.data_privacy(tree2, "node_modules/\n*.log")
|
||||
items2 = {it["name"]: it for it in dp2["items"]}
|
||||
check(items2["数据目录入库"]["pass"] is False, "data dir tracked -> fail")
|
||||
check(items2[".env 入库"]["pass"] is False, ".env tracked -> fail")
|
||||
check(items2[".gitignore 忽略 .env"]["pass"] is False, "gitignore missing .env")
|
||||
check(len(dp2["risks"]) >= 1, "risky repo -> has risks")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 6) 打分 _score_10
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_score():
|
||||
print("\n== test_score ==")
|
||||
# 5 项全 2 分 → 10
|
||||
full = [{"score": 2}] * 5
|
||||
check_eq(repro._score_10(full), 10.0, "all pass -> 10")
|
||||
# 5 项全 0 → 0
|
||||
zero = [{"score": 0}] * 5
|
||||
check_eq(repro._score_10(zero), 0.0, "all fail -> 0")
|
||||
# 混合:5 项中 3×2 + 2×0 = 6/10 = 6.0
|
||||
mixed = [{"score": 2}, {"score": 2}, {"score": 2}, {"score": 0}, {"score": 0}]
|
||||
check_eq(repro._score_10(mixed), 6.0, "mixed 6/10 -> 6.0")
|
||||
# 空列表
|
||||
check_eq(repro._score_10([]), 0.0, "empty -> 0")
|
||||
# cap 不超 10
|
||||
over = [{"score": 2}] * 8
|
||||
check(repro._score_10(over) <= 10.0, "capped at 10")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 7) render_report 端到端(用 mock 结果)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_render_report():
|
||||
print("\n== test_render_report ==")
|
||||
mock = {
|
||||
"scenario": "S3_compliance_reproducibility",
|
||||
"repo": "o/r",
|
||||
"default_branch": "master",
|
||||
"license": "MIT",
|
||||
"repro_items": [{"name": "CI 配置", "pass": True, "score": 2, "evidence": "ci"}],
|
||||
"compliance_items": [{"name": "LICENSE 文件", "pass": True, "score": 2, "evidence": "MIT"}],
|
||||
"privacy_items": [{"name": ".env 入库", "pass": True, "score": 2, "evidence": "ok"}],
|
||||
"secrets": [],
|
||||
"risks": [],
|
||||
"repro_score": 10.0,
|
||||
"compliance_score": 10.0,
|
||||
"meta": {"tree_size": 5},
|
||||
}
|
||||
md = repro.render_report(mock)
|
||||
check("# 科研项目合规与复现性检查报告" in md, "report has title")
|
||||
check("MIT" in md, "report shows license")
|
||||
check("10.0/10" in md, "report shows scores")
|
||||
check("复现性检查清单" in md, "report has repro checklist")
|
||||
check("风险项" in md, "report has risk section")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def main():
|
||||
test_identify_license()
|
||||
test_scan_secrets()
|
||||
test_repro_checks()
|
||||
test_compliance_items()
|
||||
test_data_privacy()
|
||||
test_score()
|
||||
test_render_report()
|
||||
|
||||
print()
|
||||
if _failures:
|
||||
print(f"RESULT: FAIL ({len(_failures)} failures)")
|
||||
for m in _failures:
|
||||
print(f" - {m}")
|
||||
sys.exit(1)
|
||||
else:
|
||||
print("RESULT: ALL PASS")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
@ -0,0 +1,406 @@
|
|||
"""test_visual.py — S6 科研成果可视化的纯单元测试(不联网、不调 gitlink-cli)。
|
||||
|
||||
把「取数」与「算法」分离:算法函数接收已构造好的 Python 数据结构(mock commits/issues/...),
|
||||
测试只覆盖 bin_weekly / contribution_heatmap / extract_paper_links / classify_artifacts,
|
||||
不测 plotly 渲染。
|
||||
|
||||
运行:`python scripts/research/test_visual.py`
|
||||
"""
|
||||
import os
|
||||
import sys
|
||||
from datetime import datetime, timedelta, timezone
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
import visual as V # noqa: E402
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 辅助:构造「最近 N 天」的时间戳/ISO 字符串
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _days_ago_ts(days: int) -> float:
|
||||
return (datetime.now(timezone.utc) - timedelta(days=days)).timestamp()
|
||||
|
||||
|
||||
def _days_ago_iso(days: int) -> str:
|
||||
return (datetime.now(timezone.utc) - timedelta(days=days)).strftime("%Y-%m-%dT%H:%M:%SZ")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# _to_timestamp
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_to_timestamp_int_seconds():
|
||||
assert V._to_timestamp(0) == 0.0
|
||||
assert V._to_timestamp(1719500000) == 1719500000.0
|
||||
assert V._to_timestamp("1719500000") == 1719500000.0
|
||||
|
||||
|
||||
def test_to_timestamp_millis():
|
||||
# 13 位 → 视为毫秒
|
||||
assert abs(V._to_timestamp(1719500000000) - 1719500000.0) < 1e-3
|
||||
|
||||
|
||||
def test_to_timestamp_iso():
|
||||
ts = V._to_timestamp("2024-06-15T12:00:00Z")
|
||||
assert ts > 0
|
||||
# 无效字符串 → 0
|
||||
assert V._to_timestamp("not-a-date") == 0.0
|
||||
assert V._to_timestamp(None) == 0.0
|
||||
assert V._to_timestamp("") == 0.0
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# bin_weekly
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_bin_weekly_shape_and_sum():
|
||||
weeks = 4
|
||||
commits = [
|
||||
{"timestamp": _days_ago_iso(2)}, # 本周
|
||||
{"timestamp": _days_ago_iso(9)}, # 上周
|
||||
{"timestamp": _days_ago_iso(9)},
|
||||
{"timestamp": _days_ago_iso(20)}, # 3 周前
|
||||
]
|
||||
out = V.bin_weekly(commits, weeks, V._commit_time)
|
||||
assert set(out.keys()) == {"labels", "counts"}
|
||||
assert len(out["labels"]) == weeks
|
||||
assert len(out["counts"]) == weeks
|
||||
assert sum(out["counts"]) == len(commits)
|
||||
# 标签是 ISO 周(形如 2026-Wxx)
|
||||
assert all("-W" in lab for lab in out["labels"])
|
||||
|
||||
|
||||
def test_bin_weekly_drops_out_of_window_and_invalid():
|
||||
weeks = 3
|
||||
commits = [
|
||||
{"timestamp": _days_ago_iso(1)}, # 窗口内
|
||||
{"timestamp": _days_ago_iso(365)}, # 太早(窗口外)→ 丢弃
|
||||
{"timestamp": "garbage"}, # 非法 → 丢弃
|
||||
{}, # 无时间字段 → 丢弃
|
||||
]
|
||||
out = V.bin_weekly(commits, weeks, V._commit_time)
|
||||
assert sum(out["counts"]) == 1
|
||||
|
||||
|
||||
def test_bin_weekly_issues_uses_created_at():
|
||||
weeks = 2
|
||||
issues = [
|
||||
{"created_at": _days_ago_iso(3)}, # 窗口内
|
||||
{"created_at": _days_ago_iso(40)}, # 窗口外(>2 周)→ 丢弃
|
||||
]
|
||||
out = V.bin_weekly(issues, weeks, V._issue_time)
|
||||
assert sum(out["counts"]) == 1 # 只有 3 天前那条进窗口
|
||||
|
||||
|
||||
def test_bin_weekly_prs_uses_pr_created_unix():
|
||||
weeks = 2
|
||||
prs = [
|
||||
{"pr_created_unix": int(_days_ago_ts(2))},
|
||||
{"pr_created_unix": int(_days_ago_ts(50))},
|
||||
]
|
||||
out = V.bin_weekly(prs, weeks, V._pr_time)
|
||||
assert sum(out["counts"]) == 1
|
||||
|
||||
|
||||
def test_bin_weekly_empty():
|
||||
out = V.bin_weekly([], 5, V._commit_time)
|
||||
assert len(out["labels"]) == 5
|
||||
assert out["counts"] == [0, 0, 0, 0, 0]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# contribution_heatmap
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_heatmap_matrix_shape():
|
||||
weeks = 3
|
||||
contributors = [
|
||||
{"login": "alice", "contributions": 100},
|
||||
{"login": "bob", "contributions": 5},
|
||||
]
|
||||
commits = [
|
||||
{"timestamp": _days_ago_iso(2), "author": {"login": "alice"}},
|
||||
{"timestamp": _days_ago_iso(2), "author": {"login": "alice"}},
|
||||
{"timestamp": _days_ago_iso(9), "author": {"login": "bob"}},
|
||||
]
|
||||
hm = V.contribution_heatmap(contributors, commits, weeks)
|
||||
assert set(hm.keys()) == {"users", "weeks", "matrix"}
|
||||
assert hm["users"][:2] == ["alice", "bob"]
|
||||
assert len(hm["matrix"]) == len(hm["users"])
|
||||
for row in hm["matrix"]:
|
||||
assert len(row) == weeks
|
||||
# alice 行总和 = 2
|
||||
alice_row = hm["matrix"][hm["users"].index("alice")]
|
||||
assert sum(alice_row) == 2
|
||||
bob_row = hm["matrix"][hm["users"].index("bob")]
|
||||
assert sum(bob_row) == 1
|
||||
|
||||
|
||||
def test_heatmap_includes_commit_authors_not_in_contributors():
|
||||
weeks = 2
|
||||
contributors = [{"login": "alice", "contributions": 1}]
|
||||
commits = [
|
||||
{"timestamp": _days_ago_iso(2), "author": {"login": "alice"}},
|
||||
{"timestamp": _days_ago_iso(3), "author": {"login": "carol"}}, # 不在 contributors
|
||||
]
|
||||
hm = V.contribution_heatmap(contributors, commits, weeks)
|
||||
assert "carol" in hm["users"]
|
||||
|
||||
|
||||
def test_heatmap_top_users_cap():
|
||||
weeks = 2
|
||||
contributors = [{"login": f"u{i}", "contributions": i} for i in range(20)]
|
||||
commits = [{"timestamp": _days_ago_iso(1), "author": {"login": f"u{i}"}}
|
||||
for i in range(20)]
|
||||
hm = V.contribution_heatmap(contributors, commits, weeks, top_users=5)
|
||||
assert len(hm["users"]) <= 5
|
||||
assert len(hm["matrix"]) == len(hm["users"])
|
||||
|
||||
|
||||
def test_heatmap_empty():
|
||||
hm = V.contribution_heatmap([], [], 4)
|
||||
assert hm["users"] == []
|
||||
assert hm["matrix"] == []
|
||||
assert len(hm["weeks"]) == 4
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# extract_paper_links
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_extract_arxiv_url():
|
||||
text = "See https://arxiv.org/abs/2401.00012 for details."
|
||||
links = V.extract_paper_links(text)
|
||||
assert len(links) == 1
|
||||
assert links[0]["type"] == "arxiv"
|
||||
assert links[0]["target"] == "https://arxiv.org/abs/2401.00012"
|
||||
# snippet 截取自原文上下文(窗口较窄,断言前缀即可)
|
||||
assert links[0]["source_text_snippet"].startswith("See https://arxiv.org")
|
||||
|
||||
|
||||
def test_extract_arxiv_pdf_url():
|
||||
text = "paper: https://arxiv.org/pdf/2305.12345.pdf"
|
||||
links = V.extract_paper_links(text)
|
||||
assert len(links) == 1
|
||||
# 归一成 abs 形式
|
||||
assert links[0]["target"] == "https://arxiv.org/abs/2305.12345"
|
||||
|
||||
|
||||
def test_extract_arxiv_bare():
|
||||
text = "We use arXiv:2103.07018 in our method."
|
||||
links = V.extract_paper_links(text)
|
||||
assert any(l["target"] == "https://arxiv.org/abs/2103.07018" for l in links)
|
||||
|
||||
|
||||
def test_extract_doi_url():
|
||||
text = "Cited from https://doi.org/10.1000/182"
|
||||
links = V.extract_paper_links(text)
|
||||
assert len(links) == 1
|
||||
assert links[0]["type"] == "doi"
|
||||
assert links[0]["target"] == "https://doi.org/10.1000/182"
|
||||
|
||||
|
||||
def test_extract_doi_bare():
|
||||
text = "Reference 10.1109/5.771073 shows that."
|
||||
links = V.extract_paper_links(text)
|
||||
assert len(links) == 1
|
||||
assert links[0]["target"] == "https://doi.org/10.1109/5.771073"
|
||||
|
||||
|
||||
def test_extract_dedup_same_id():
|
||||
text = ("arxiv 1 https://arxiv.org/abs/2401.00012 "
|
||||
"and again https://arxiv.org/abs/2401.00012")
|
||||
links = V.extract_paper_links(text)
|
||||
assert len(links) == 1
|
||||
|
||||
|
||||
def test_extract_dedup_across_doi_forms():
|
||||
# doi.org 形式与裸 DOI 视为同一条
|
||||
text = "https://doi.org/10.1000/182 and bare 10.1000/182 again"
|
||||
links = V.extract_paper_links(text)
|
||||
# 同一 DOI 只出现一次
|
||||
targets = [l["target"] for l in links]
|
||||
assert targets.count("https://doi.org/10.1000/182") == 1
|
||||
|
||||
|
||||
def test_extract_multiple_and_order():
|
||||
text = ("first https://arxiv.org/abs/2401.00012 "
|
||||
"then https://doi.org/10.1000/182")
|
||||
links = V.extract_paper_links(text)
|
||||
assert len(links) == 2
|
||||
# 按出现位置排序
|
||||
assert links[0]["type"] == "arxiv"
|
||||
assert links[1]["type"] == "doi"
|
||||
|
||||
|
||||
def test_extract_none_in_text():
|
||||
assert V.extract_paper_links("no links here at all") == []
|
||||
assert V.extract_paper_links("") == []
|
||||
|
||||
|
||||
def test_extract_strips_trailing_punct_from_doi():
|
||||
text = "see 10.1000/abc123, then more."
|
||||
links = V.extract_paper_links(text)
|
||||
assert links[0]["target"].endswith("/abc123") # 末尾逗号/句号被清掉
|
||||
assert not links[0]["target"].rstrip().endswith(",")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# classify_artifacts
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_classify_paper_and_ipynb():
|
||||
tree = [
|
||||
{"path": "docs/paper.pdf"},
|
||||
{"path": "notebooks/demo.ipynb"},
|
||||
]
|
||||
out = V.classify_artifacts(tree)
|
||||
cats = {a["path"]: a["category"] for a in out}
|
||||
assert cats["docs/paper.pdf"] == "paper"
|
||||
assert cats["notebooks/demo.ipynb"] == "paper"
|
||||
|
||||
|
||||
def test_classify_dataset():
|
||||
tree = [
|
||||
{"path": "data/train.csv"},
|
||||
{"path": "datasets/x.parquet"},
|
||||
]
|
||||
out = V.classify_artifacts(tree)
|
||||
cats = {a["path"]: a["category"] for a in out}
|
||||
assert cats["data/train.csv"] == "dataset"
|
||||
|
||||
|
||||
def test_classify_model():
|
||||
tree = [
|
||||
{"path": "model/best.ckpt"},
|
||||
{"path": "models/v2.onnx"},
|
||||
]
|
||||
out = V.classify_artifacts(tree)
|
||||
cats = {a["path"]: a["category"] for a in out}
|
||||
assert cats["model/best.ckpt"] == "model"
|
||||
assert cats["models/v2.onnx"] == "model"
|
||||
|
||||
|
||||
def test_classify_benchmark():
|
||||
tree = [{"path": "benchmark/glue/run.py"}]
|
||||
out = V.classify_artifacts(tree)
|
||||
assert out[0]["category"] == "benchmark"
|
||||
|
||||
|
||||
def test_classify_ignores_unrelated():
|
||||
tree = [
|
||||
{"path": "src/main.py"},
|
||||
{"path": "README.md"},
|
||||
{"path": "tools/util.go"},
|
||||
]
|
||||
out = V.classify_artifacts(tree)
|
||||
assert out == [] # 都不命中任何类别
|
||||
|
||||
|
||||
def test_classify_dedup_same_path():
|
||||
tree = [
|
||||
{"path": "data/a.csv"},
|
||||
{"path": "data/a.csv"}, # 重复
|
||||
]
|
||||
out = V.classify_artifacts(tree)
|
||||
assert len(out) == 1
|
||||
|
||||
|
||||
def test_classify_handles_name_only():
|
||||
# 没有 path 只有 name 的条目也能处理
|
||||
tree = [{"name": "paper.pdf"}]
|
||||
out = V.classify_artifacts(tree)
|
||||
assert len(out) == 1
|
||||
assert out[0]["category"] == "paper"
|
||||
|
||||
|
||||
def test_classify_empty_and_non_dict():
|
||||
assert V.classify_artifacts([]) == []
|
||||
assert V.classify_artifacts(None) == []
|
||||
assert V.classify_artifacts(["str", 123, None]) == []
|
||||
|
||||
|
||||
def test_artifact_summary():
|
||||
arts = [
|
||||
{"path": "a.pdf", "category": "paper"},
|
||||
{"path": "b.ipynb", "category": "paper"},
|
||||
{"path": "x.csv", "category": "dataset"},
|
||||
{"path": "m.ckpt", "category": "model"},
|
||||
]
|
||||
s = V.artifact_summary(arts)
|
||||
assert s == {"paper": 2, "dataset": 1, "model": 1, "benchmark": 0}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# render_report(不渲染 plotly,只验证 markdown 结构)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_render_report_has_sections():
|
||||
result = {
|
||||
"repo": "o/r", "weeks": 4,
|
||||
"timeline": {"labels": ["W1", "W2", "W3", "W4"],
|
||||
"commits": [1, 2, 3, 4], "issues": [0, 1, 0, 2],
|
||||
"prs": [0, 0, 1, 0]},
|
||||
"heatmap": {"users": ["alice", "bob"], "weeks": ["W1", "W2"],
|
||||
"matrix": [[1, 2], [0, 1]]},
|
||||
"languages": {"Python": "99%"},
|
||||
"milestones": [],
|
||||
"paper_links": [{"type": "arxiv", "target": "https://arxiv.org/abs/2401.00012",
|
||||
"source_text_snippet": "see arxiv"}],
|
||||
"artifacts": [{"path": "p.pdf", "category": "paper", "name": "p.pdf"}],
|
||||
"artifact_summary": {"paper": 1, "dataset": 0, "model": 0, "benchmark": 0},
|
||||
"meta": {"commit_count": 10, "issue_count": 3, "pr_count": 1,
|
||||
"contributor_count": 2, "milestone_count": 0},
|
||||
}
|
||||
md = V.render_report(result)
|
||||
assert "科研成果可视化" in md
|
||||
assert "o/r" in md
|
||||
assert "alice" in md
|
||||
assert "arxiv.org/abs/2401.00012" in md
|
||||
# 含周快照表头
|
||||
assert "commits" in md
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 端到端(算法层):run() 复用 raw dict,不联网
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_run_with_mock_raw():
|
||||
raw = {
|
||||
"commits": [{"timestamp": _days_ago_iso(2), "author": {"login": "alice"},
|
||||
"message": "see https://arxiv.org/abs/2401.00012"}],
|
||||
"issues": [{"created_at": _days_ago_iso(3)}],
|
||||
"prs": [{"pr_created_unix": int(_days_ago_ts(4))}],
|
||||
"milestones": [{"name": "v1.0", "due_on": _days_ago_iso(30)}],
|
||||
"languages": {"Python": "99%"},
|
||||
"contributors": [{"login": "alice", "contributions": 1}],
|
||||
"readme": "ref https://doi.org/10.1000/182 here",
|
||||
"tree": [{"path": "data/x.csv"}, {"path": "paper.pdf"}],
|
||||
}
|
||||
result = V.run("owner", "repo", weeks=4, raw=raw)
|
||||
assert result["scenario"] == "S6_research_visualization"
|
||||
assert result["repo"] == "owner/repo"
|
||||
assert result["weeks"] == 4
|
||||
# 论文链接同时来自 readme 和 commit message
|
||||
targets = {p["target"] for p in result["paper_links"]}
|
||||
assert "https://arxiv.org/abs/2401.00012" in targets
|
||||
assert "https://doi.org/10.1000/182" in targets
|
||||
# 产物分类
|
||||
cats = {a["category"] for a in result["artifacts"]}
|
||||
assert cats == {"dataset", "paper"}
|
||||
# 时间线长度 = weeks
|
||||
assert len(result["timeline"]["labels"]) == 4
|
||||
|
||||
|
||||
def _run_all():
|
||||
fns = [v for k, v in sorted(globals().items()) if k.startswith("test_")]
|
||||
for fn in fns:
|
||||
fn()
|
||||
print(f"PASS {fn.__name__}")
|
||||
print(f"\nAll {len(fns)} visual tests passed.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
_run_all()
|
||||
|
|
@ -0,0 +1,110 @@
|
|||
"""topics.py — 科研主题/技术关键词词典(S2 知识图谱与 S4 协作匹配共享)。
|
||||
|
||||
采用「关键词词典 + 字符串匹配」方式抽取主题,零 NLP/分词依赖,结果确定可复现。
|
||||
词典可扩展;覆盖 GitLink 上常见科研方向(CV/NLP/RL/DL/系统/安全/科学计算等)。
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from collections import Counter
|
||||
from typing import Iterable
|
||||
|
||||
# 主题 → 触发关键词(中英文)。小写匹配。
|
||||
TOPIC_KEYWORDS: dict[str, tuple[str, ...]] = {
|
||||
"machine_learning": ("机器学习", "machine learning", "监督学习", "无监督学习",
|
||||
"supervised", "unsupervised", "scikit-learn", "sklearn",
|
||||
"特征工程", "feature engineering", "generalization"),
|
||||
"computer_vision": ("目标检测", "图像分类", "语义分割", "实例分割", "目标跟踪",
|
||||
"object detection", "image classification", "semantic segmentation",
|
||||
"instance segmentation", "object tracking", "yolo", "resnet", "cnn",
|
||||
"图像识别", "ocr", "人脸识别", "visual", "vision"),
|
||||
"nlp": ("自然语言处理", "文本分类", "机器翻译", "问答", "命名实体",
|
||||
"nlp", "text classification", "machine translation", "transformer", "bert",
|
||||
"gpt", "llm", "大模型", "大语言模型", "预训练", "pretrain", "分词", "tokeniz"),
|
||||
"generative_ai": ("生成式", "aigc", "扩散模型", "生成模型", "文生图", "多模态",
|
||||
"generative", "diffusion", "gan", "vae", "multimodal", "clip",
|
||||
"对话", "chatgpt", "chat", "instruction tun"),
|
||||
"reinforcement_learning": ("强化学习", "多智能体", "决策",
|
||||
"reinforcement learning", "multi-agent", "ppo", "dqn",
|
||||
"q-learning", "reward", "policy gradient"),
|
||||
"deep_learning": ("深度学习", "神经网络", "训练", "推理", "微调",
|
||||
"deep learning", "neural network", "pytorch", "tensorflow",
|
||||
"mindspore", "paddle", "paddlepaddle", "inference", "fine-tun",
|
||||
"backbone", "checkpoint"),
|
||||
"graph_learning": ("图神经网络", "图表示学习", "知识图谱",
|
||||
"graph neural", "gnn", "graph convolution", "gcn", "graphsage",
|
||||
"knowledge graph", "图嵌入", "graph embed"),
|
||||
"federated_learning": ("联邦学习", "隐私保护", "分布式训练",
|
||||
"federated", "privacy", "distributed training"),
|
||||
"speech": ("语音识别", "语音合成", "声纹", "语音",
|
||||
"speech", "asr", "tts", "speaker", "voice", "声学"),
|
||||
"scientific_computing": ("科学计算", "数值模拟", "高性能计算", "并行计算",
|
||||
"numerical", "simulation", "hpc", "parallel", "cuda", "gpu",
|
||||
"有限元", "偏微分"),
|
||||
"autonomous_systems": ("自动驾驶", "机器人", "感知", "导航", "slam",
|
||||
"autonomous", "robotics", "robot", "self-driving", "planning"),
|
||||
"bioinformatics": ("生物信息", "蛋白质", "基因", "分子",
|
||||
"bioinformatic", "genomic", "protein", "molecular", "drug"),
|
||||
"time_series": ("时序", "时间序列", "预测", "序列建模",
|
||||
"time series", "time-series", "forecasting", "temporal"),
|
||||
"devops": ("ci/cd", "devops", "pipeline", "容器", "编排",
|
||||
"docker", "kubernetes", "k8s", "jenkins", "自动化部署", "helm"),
|
||||
"database": ("数据库", "存储", "索引",
|
||||
"database", "sql", "nosql", "storage", "index"),
|
||||
"security": ("安全", "漏洞", "加密", "隐私",
|
||||
"security", "vulnerability", "crypto", "privacy", "attack"),
|
||||
"data_mining": ("数据挖掘", "推荐系统", "聚类", "分类",
|
||||
"data mining", "recommender", "clustering", "classification", "tf-idf"),
|
||||
}
|
||||
|
||||
# 编程语言关键词(用于 S4 语言匹配)
|
||||
LANGUAGE_KEYWORDS: tuple[str, ...] = (
|
||||
"python", "go", "golang", "c++", "cpp", "c#", "java", "rust", "javascript",
|
||||
"typescript", "julia", "r", "matlab", "scala", "swift", "kotlin", "cuda",
|
||||
)
|
||||
|
||||
_NON_ALNUM = re.compile(r"[^\w一-鿿+#]+")
|
||||
|
||||
|
||||
def _normalize(text: str) -> str:
|
||||
return (text or "").lower()
|
||||
|
||||
|
||||
def extract_topics(text: str) -> list[str]:
|
||||
"""从一段文本中抽取命中的主题列表(去重,保序)。"""
|
||||
t = _normalize(text)
|
||||
if not t:
|
||||
return []
|
||||
hit = []
|
||||
for topic, kws in TOPIC_KEYWORDS.items():
|
||||
for kw in kws:
|
||||
if _normalize(kw) in t:
|
||||
hit.append(topic)
|
||||
break
|
||||
return hit
|
||||
|
||||
|
||||
def extract_languages(text: str) -> list[str]:
|
||||
"""从文本中抽取命中的编程语言(归一化别名,如 golang→go)。"""
|
||||
t = _normalize(text)
|
||||
if not t:
|
||||
return []
|
||||
alias = {"golang": "go", "cpp": "c++", "c#": "c#", "ts": "typescript"}
|
||||
out, seen = [], set()
|
||||
# 按非字母数字分割后逐 token 比对,避免 'go' 误命中 'google'
|
||||
tokens = set(_NON_ALNUM.sub(" ", t).split())
|
||||
for kw in LANGUAGE_KEYWORDS:
|
||||
norm = _normalize(kw)
|
||||
if norm in tokens and norm not in seen:
|
||||
seen.add(norm)
|
||||
out.append(alias.get(norm, norm))
|
||||
return out
|
||||
|
||||
|
||||
def topic_counter(texts: Iterable[str]) -> Counter:
|
||||
"""对多段文本累计主题词频,用于热点排序(S2)。"""
|
||||
c: Counter = Counter()
|
||||
for t in texts:
|
||||
for topic in extract_topics(t):
|
||||
c[topic] += 1
|
||||
return c
|
||||
|
|
@ -0,0 +1,636 @@
|
|||
"""visual.py — S6 科研成果可视化沉淀。
|
||||
|
||||
把一个科研仓库的「成果」沉淀成一张可交互的可视化报告:开发时间线(commit/issue/pr 周粒度
|
||||
趋势)、贡献者×周热力图、语言占比饼图、里程碑甘特。同时从 README/提交里抽取论文引用
|
||||
(arXiv / DOI)并按目录对仓库产物做分类,便于科研工作者一眼看清「成果产出节奏 + 引用源头」。
|
||||
|
||||
数据全部经 gitlink-cli 获取(commit via Raw API、issue/pr/milestone/repo +list、contributors、
|
||||
languages、readme、tree)。算法纯函数化(按周分桶 / 热力矩阵 / 论文链接抽取 / 产物分类),
|
||||
便于离线单测。
|
||||
|
||||
用法:
|
||||
python visual.py --owner mindspore-Ecosystem --repo mindspore --weeks 26 --out ./out
|
||||
python visual.py --owner O --repo R # 仅打印 JSON
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
import collect as c # noqa: E402
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 时间解析工具
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _to_timestamp(value: Any) -> float:
|
||||
"""把 GitLink 多种时间表示统一成 epoch 秒。
|
||||
|
||||
支持整数秒(pr_created_unix)、ISO 字符串(created_at / timestamp 字符串)。
|
||||
无法解析返回 0.0(最远古时间,会被周分桶丢弃到「太早」一端)。
|
||||
"""
|
||||
if value is None:
|
||||
return 0.0
|
||||
# 整数秒(commit.timestamp 形如 "1719500000" 也可走这里)
|
||||
if isinstance(value, (int, float)):
|
||||
f = float(value)
|
||||
# 毫秒级时间戳兜底(13 位)
|
||||
return f / 1000.0 if f > 1e12 else f
|
||||
s = str(value).strip()
|
||||
if not s:
|
||||
return 0.0
|
||||
# 纯数字字符串
|
||||
if re.fullmatch(r"\d+(\.\d+)?", s):
|
||||
f = float(s)
|
||||
return f / 1000.0 if f > 1e12 else f
|
||||
# ISO 8601(兼容带/不带 Z、带毫秒、带时区偏移)
|
||||
txt = s.replace("Z", "+00:00")
|
||||
fmts = ("%Y-%m-%dT%H:%M:%S%z",
|
||||
"%Y-%m-%dT%H:%M:%S.%f%z",
|
||||
"%Y-%m-%d %H:%M:%S",
|
||||
"%Y-%m-%d")
|
||||
for fmt in fmts:
|
||||
try:
|
||||
dt = datetime.strptime(txt, fmt)
|
||||
if dt.tzinfo is None:
|
||||
dt = dt.replace(tzinfo=timezone.utc)
|
||||
return dt.timestamp()
|
||||
except ValueError:
|
||||
continue
|
||||
return 0.0
|
||||
|
||||
|
||||
def _commit_time(commit: dict) -> float:
|
||||
"""提交对象取时间戳:优先 timestamp(字符串),退而 author.committed_unix。"""
|
||||
ts = commit.get("timestamp")
|
||||
if ts is not None:
|
||||
return _to_timestamp(ts)
|
||||
auth = commit.get("author") or {}
|
||||
if isinstance(auth, dict):
|
||||
for k in ("committed_unix", "committed_at", "time", "date"):
|
||||
if auth.get(k) is not None:
|
||||
return _to_timestamp(auth.get(k))
|
||||
return 0.0
|
||||
|
||||
|
||||
def _issue_time(issue: dict) -> float:
|
||||
return _to_timestamp(issue.get("created_at"))
|
||||
|
||||
|
||||
def _pr_time(pr: dict) -> float:
|
||||
return _to_timestamp(pr.get("pr_created_unix") or pr.get("created_at"))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 算法 1:按周分桶
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def bin_weekly(items: list, weeks: int, time_getter) -> dict[str, list]:
|
||||
"""把带时间戳的对象按「最近 weeks 周」分桶(含本周在内的 weeks 个连续周桶)。
|
||||
|
||||
Args:
|
||||
items: 待分桶对象列表。
|
||||
weeks: 保留最近多少个周桶。
|
||||
time_getter: 从单个对象取 epoch 秒的函数。
|
||||
|
||||
返回 ``{"labels": [...], "counts": [...]}``:
|
||||
- labels[i] 形如 "2026-W13"(ISO 周标签),从最近一周倒序到最老一周。
|
||||
- counts[i] 为该周命中数;时间非法或越界(早于窗口左端)的对象不计入。
|
||||
"""
|
||||
weeks = max(1, int(weeks))
|
||||
now = datetime.now(timezone.utc)
|
||||
# 右端边界对齐到「下周一 00:00 UTC」(不含),使窗口包含当前周在内共 weeks 周。
|
||||
# 若右端用「本周一」,当前周会被整体排除,最近一周的数据被吞掉。
|
||||
today = now.replace(hour=0, minute=0, second=0, microsecond=0)
|
||||
# Monday=0 .. Sunday=6;toordinal - weekday = 本周一,+7 = 下周一
|
||||
next_monday = today.fromordinal(today.toordinal() - today.weekday() + 7).replace(tzinfo=timezone.utc)
|
||||
end_ts = next_monday.timestamp()
|
||||
start_ts = end_ts - weeks * 7 * 86400
|
||||
|
||||
counts = [0] * weeks
|
||||
iso_labels: list[str] = []
|
||||
for i in range(weeks):
|
||||
# 桶 i 的周一 = end - (weeks-1-i) 周
|
||||
bucket_monday_ts = start_ts + i * 7 * 86400
|
||||
bucket_monday = datetime.fromtimestamp(bucket_monday_ts, tz=timezone.utc)
|
||||
iso_year, iso_week, _ = bucket_monday.isocalendar()
|
||||
iso_labels.append(f"{iso_year}-W{iso_week:02d}")
|
||||
|
||||
for item in items:
|
||||
ts = time_getter(item)
|
||||
if ts <= 0:
|
||||
continue
|
||||
if ts < start_ts or ts >= end_ts:
|
||||
continue
|
||||
offset = ts - start_ts
|
||||
idx = int(offset // (7 * 86400))
|
||||
if 0 <= idx < weeks:
|
||||
counts[idx] += 1
|
||||
|
||||
return {"labels": iso_labels, "counts": counts}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 算法 2:贡献者 × 周热力矩阵
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def contribution_heatmap(contributors: list, commits: list, weeks: int,
|
||||
top_users: int = 12) -> dict[str, list]:
|
||||
"""构建 top 贡献者 × 周桶的提交数矩阵。
|
||||
|
||||
Args:
|
||||
contributors: GitLink contributors[](用于排序与展示名)。
|
||||
commits: GitLink commits[](含 author.login)。
|
||||
weeks: 周桶数(与 bin_weekly 同口径)。
|
||||
top_users: 矩阵最多保留多少个贡献者(按贡献数 desc)。
|
||||
|
||||
返回 ``{"users": [login...], "weeks": [label...], "matrix": [[cnt...]]}``:
|
||||
matrix[user_i][week_j] = 该用户在该周的提交数。无 contributor 信息时也按 commit 作者聚合。
|
||||
"""
|
||||
weeks = max(1, int(weeks))
|
||||
# 用 bin_weekly 的同口径周标签(取贡献数排序后的 login 列表)
|
||||
window = bin_weekly(commits, weeks, _commit_time)
|
||||
week_labels = window["labels"]
|
||||
now = datetime.now(timezone.utc)
|
||||
end = now.replace(hour=0, minute=0, second=0, microsecond=0)
|
||||
end = end.fromordinal(end.toordinal() - end.weekday()).replace(tzinfo=timezone.utc)
|
||||
start_ts = end.timestamp() - weeks * 7 * 86400
|
||||
|
||||
# 候选用户顺序:contributors(按 contributions desc)+ 提交里出现但不在 contributors 的作者
|
||||
ordered: list[str] = []
|
||||
seen: set[str] = set()
|
||||
for contrib in contributors or []:
|
||||
login = c.login_of(contrib) or ""
|
||||
if login and login not in seen:
|
||||
ordered.append(login)
|
||||
seen.add(login)
|
||||
for cm in commits or []:
|
||||
auth = cm.get("author") or {}
|
||||
login = c.login_of(auth) if isinstance(auth, dict) else ""
|
||||
if login and login not in seen:
|
||||
ordered.append(login)
|
||||
seen.add(login)
|
||||
|
||||
users = ordered[:top_users]
|
||||
matrix = [[0] * weeks for _ in users]
|
||||
user_idx = {u: i for i, u in enumerate(users)}
|
||||
|
||||
for cm in commits or []:
|
||||
ts = _commit_time(cm)
|
||||
if ts <= 0 or ts < start_ts or ts >= end.timestamp():
|
||||
continue
|
||||
auth = cm.get("author") or {}
|
||||
login = c.login_of(auth) if isinstance(auth, dict) else ""
|
||||
if not login or login not in user_idx:
|
||||
continue
|
||||
offset = ts - start_ts
|
||||
idx = int(offset // (7 * 86400))
|
||||
if 0 <= idx < weeks:
|
||||
matrix[user_idx[login]][idx] += 1
|
||||
|
||||
return {"users": users, "weeks": week_labels, "matrix": matrix}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 算法 3:论文引用链接抽取
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# arXiv: arxiv.org/abs/2401.00012 / arxiv.org/pdf/... / arxiv:2401.00012
|
||||
_ARXIV_RE = re.compile(
|
||||
r"(?:https?://)?(?:www\.)?arxiv\.org/(?:abs|pdf)/(\d{4}\.\d{4,5})(?:v\d+)?(?:\.pdf)?",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_ARXIV_BARE_RE = re.compile(r"\barXiv:\s*(\d{4}\.\d{4,5})", re.IGNORECASE)
|
||||
# DOI: doi.org/10.xxxx/... 或裸 10.xxxx/...(论文里的 DOI 形式)
|
||||
_DOI_URL_RE = re.compile(
|
||||
r"(?:https?://)?(?:dx\.)?doi\.org/(10\.\d{4,9}/[^\s)\"'<>]+)", re.IGNORECASE,
|
||||
)
|
||||
_DOI_BARE_RE = re.compile(
|
||||
r"\b(10\.\d{4,9}/[^\s)\"'<>]+)", re.IGNORECASE,
|
||||
)
|
||||
|
||||
# 在 DOI 字符串里清掉常见尾部分隔符(避免吃进句号、逗号)
|
||||
_TRAILING_PUNCT = ".,;:)\"'>]"
|
||||
|
||||
|
||||
def _clean_doi(doi: str) -> str:
|
||||
return doi.rstrip(_TRAILING_PUNCT)
|
||||
|
||||
|
||||
def _snippet(text: str, pos: int, span: int = 60) -> str:
|
||||
"""以匹配位置为中心截一段上下文。"""
|
||||
a = max(0, pos - span // 2)
|
||||
b = min(len(text), pos + span // 2)
|
||||
frag = text[a:b].replace("\n", " ").strip()
|
||||
return ("…" + frag) if a > 0 else frag
|
||||
|
||||
|
||||
def extract_paper_links(text: str) -> list[dict[str, str]]:
|
||||
"""从文本里抽取 arXiv 与 DOI 论文引用链接。
|
||||
|
||||
返回 ``[{"source_text_snippet": str, "target": url, "type": "arxiv"|"doi"}]``,
|
||||
按 (出现位置, 类型优先 arxiv) 排序,去重(同一 arxiv id / doi 只保留首次)。
|
||||
"""
|
||||
if not text:
|
||||
return []
|
||||
out: list[dict[str, str]] = []
|
||||
seen_arxiv: set[str] = set()
|
||||
seen_doi: set[str] = set()
|
||||
hits: list[tuple[int, dict[str, str]]] = []
|
||||
|
||||
for m in _ARXIV_RE.finditer(text):
|
||||
aid = m.group(1)
|
||||
if aid in seen_arxiv:
|
||||
continue
|
||||
seen_arxiv.add(aid)
|
||||
hits.append((m.start(), {
|
||||
"source_text_snippet": _snippet(text, m.start()),
|
||||
"target": f"https://arxiv.org/abs/{aid}",
|
||||
"type": "arxiv",
|
||||
}))
|
||||
for m in _ARXIV_BARE_RE.finditer(text):
|
||||
aid = m.group(1)
|
||||
if aid in seen_arxiv:
|
||||
continue
|
||||
seen_arxiv.add(aid)
|
||||
hits.append((m.start(), {
|
||||
"source_text_snippet": _snippet(text, m.start()),
|
||||
"target": f"https://arxiv.org/abs/{aid}",
|
||||
"type": "arxiv",
|
||||
}))
|
||||
for m in _DOI_URL_RE.finditer(text):
|
||||
doi = _clean_doi(m.group(1))
|
||||
if doi.lower() in seen_doi:
|
||||
continue
|
||||
seen_doi.add(doi.lower())
|
||||
hits.append((m.start(), {
|
||||
"source_text_snippet": _snippet(text, m.start()),
|
||||
"target": f"https://doi.org/{doi}",
|
||||
"type": "doi",
|
||||
}))
|
||||
for m in _DOI_BARE_RE.finditer(text):
|
||||
doi = _clean_doi(m.group(1))
|
||||
if doi.lower() in seen_doi:
|
||||
continue
|
||||
seen_doi.add(doi.lower())
|
||||
hits.append((m.start(), {
|
||||
"source_text_snippet": _snippet(text, m.start()),
|
||||
"target": f"https://doi.org/{doi}",
|
||||
"type": "doi",
|
||||
}))
|
||||
|
||||
hits.sort(key=lambda x: (x[0], 0 if x[1]["type"] == "arxiv" else 1))
|
||||
return [h[1] for h in hits]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 算法 4:仓库产物分类
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def classify_artifacts(tree_entries: list) -> list[dict[str, Any]]:
|
||||
"""按路径把仓库文件/目录归入科研产物类别。
|
||||
|
||||
规则(按优先级,先匹配先归类):
|
||||
- path 含 ``benchmark/`` 段 → benchmark
|
||||
- path 含 ``model/`` 段或 *.ckpt/*.safetensors/*.onnx → model
|
||||
- path 含 ``data/`` 段或 *.csv/*.parquet → dataset
|
||||
- *.pdf / *.ipynb / paper 关键词 → paper
|
||||
|
||||
每个产物 ``{"path", "category", "name"}``。tree_entries 既可能是文件列表
|
||||
(含 path/name/type)也可能是目录项;本函数尽力取 path/name 字段。
|
||||
"""
|
||||
out: list[dict[str, Any]] = []
|
||||
seen: set[str] = set()
|
||||
for entry in tree_entries or []:
|
||||
if not isinstance(entry, dict):
|
||||
continue
|
||||
path = entry.get("path") or entry.get("name") or ""
|
||||
if not path:
|
||||
continue
|
||||
norm = path.replace("\\", "/").lower()
|
||||
name = norm.rsplit("/", 1)[-1]
|
||||
category = None
|
||||
# benchmark(必须含 benchmark 目录段,避免误把文件名含词的归入)
|
||||
if "/benchmark/" in norm or norm.startswith("benchmark/"):
|
||||
category = "benchmark"
|
||||
elif "/model/" in norm or norm.startswith("model/") or name.endswith(
|
||||
(".ckpt", ".safetensors", ".onnx", ".pb", ".h5", ".pt")):
|
||||
category = "model"
|
||||
elif "/data/" in norm or norm.startswith("data/") or name.endswith(
|
||||
(".csv", ".parquet", ".npy", ".npz", ".hdf5", ".h5")):
|
||||
# .h5 已先被 model 吃掉,这里主要 csv/parquet/npy
|
||||
category = "dataset"
|
||||
elif (name.endswith(".pdf") or name.endswith(".ipynb")
|
||||
or "paper" in norm or "arxiv" in norm):
|
||||
category = "paper"
|
||||
if category and path not in seen:
|
||||
seen.add(path)
|
||||
out.append({"path": path, "category": category,
|
||||
"name": name or path.rsplit("/", 1)[-1]})
|
||||
return out
|
||||
|
||||
|
||||
def artifact_summary(artifacts: list[dict[str, Any]]) -> dict[str, int]:
|
||||
"""统计各类产物数量,返回 {paper: n, dataset: n, model: n, benchmark: n}。"""
|
||||
summary: dict[str, int] = {"paper": 0, "dataset": 0, "model": 0, "benchmark": 0}
|
||||
for a in artifacts or []:
|
||||
cat = a.get("category")
|
||||
if cat in summary:
|
||||
summary[cat] += 1
|
||||
return summary
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 数据采集(取数层,主流程调用;单测不触达)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def collect(owner: str, repo: str, weeks: int) -> dict[str, Any]:
|
||||
"""从 GitLink 取本场景所需的全部数据。"""
|
||||
# commits 取够 ~weeks 周(每周按 30 条粗估,上限 max_pages=10)
|
||||
cm_pages = max(2, min(10, (weeks // 3) + 1))
|
||||
commits = c.commits(owner, repo, ref="master", max_pages=cm_pages, page_size=100)
|
||||
issues = c.issues_all(owner, repo, max_pages=10, page_size=50)
|
||||
pullreqs = c.prs_all(owner, repo, max_pages=10, page_size=50)
|
||||
milestones = c.milestones(owner, repo, state="all")
|
||||
langs = c.languages(owner, repo)
|
||||
contribs = c.contributors(owner, repo)
|
||||
readme = c.readme(owner, repo)
|
||||
tree = c.tree(owner, repo)
|
||||
return {
|
||||
"info": c.repo_info(owner, repo),
|
||||
"commits": commits,
|
||||
"issues": issues,
|
||||
"prs": pullreqs,
|
||||
"milestones": milestones,
|
||||
"languages": langs,
|
||||
"contributors": contribs,
|
||||
"readme": readme,
|
||||
"tree": tree,
|
||||
}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 主算法:组装结果 dict
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def run(owner: str, repo: str, weeks: int, raw: dict[str, Any] | None = None) -> dict[str, Any]:
|
||||
"""主入口:取数(或复用传入的 raw)→ 算法 → 结果 dict。"""
|
||||
if raw is None:
|
||||
raw = collect(owner, repo, weeks)
|
||||
|
||||
commits = raw.get("commits") or []
|
||||
issues = raw.get("issues") or []
|
||||
prs = raw.get("prs") or []
|
||||
milestones = raw.get("milestones") or []
|
||||
langs = raw.get("languages") or {}
|
||||
contribs = raw.get("contributors") or []
|
||||
readme = raw.get("readme") or ""
|
||||
tree = raw.get("tree") or []
|
||||
|
||||
commits_ts = bin_weekly(commits, weeks, _commit_time)
|
||||
issues_ts = bin_weekly(issues, weeks, _issue_time)
|
||||
prs_ts = bin_weekly(prs, weeks, _pr_time)
|
||||
heatmap = contribution_heatmap(contribs, commits, weeks)
|
||||
|
||||
# 合并 readme + 提交信息作为论文链接抽取语料
|
||||
corpus_parts = [readme]
|
||||
for cm in commits[:50]:
|
||||
msg = cm.get("message") or ""
|
||||
if isinstance(msg, str):
|
||||
corpus_parts.append(msg)
|
||||
paper_links = extract_paper_links("\n".join(corpus_parts))
|
||||
|
||||
artifacts = classify_artifacts(tree)
|
||||
art_summary = artifact_summary(artifacts)
|
||||
|
||||
# 里程碑甘特数据:取有 due_on 的,转成 [start, end, title]
|
||||
gantt: list[dict[str, Any]] = []
|
||||
for ms in milestones:
|
||||
if not isinstance(ms, dict):
|
||||
continue
|
||||
title = ms.get("name") or ms.get("title") or ""
|
||||
due = _to_timestamp(ms.get("due_on") or ms.get("effective_date"))
|
||||
start = _to_timestamp(ms.get("start_date"))
|
||||
if due > 0:
|
||||
gantt.append({
|
||||
"title": title,
|
||||
"start": start if start > 0 else due - 14 * 86400,
|
||||
"due": due,
|
||||
})
|
||||
|
||||
return {
|
||||
"scenario": "S6_research_visualization",
|
||||
"repo": f"{owner}/{repo}",
|
||||
"weeks": weeks,
|
||||
"timeline": {
|
||||
"labels": commits_ts["labels"],
|
||||
"commits": commits_ts["counts"],
|
||||
"issues": issues_ts["counts"],
|
||||
"prs": prs_ts["counts"],
|
||||
},
|
||||
"heatmap": heatmap,
|
||||
"languages": langs,
|
||||
"milestones": gantt,
|
||||
"paper_links": paper_links,
|
||||
"artifacts": artifacts,
|
||||
"artifact_summary": art_summary,
|
||||
"meta": {
|
||||
"commit_count": len(commits),
|
||||
"issue_count": len(issues),
|
||||
"pr_count": len(prs),
|
||||
"milestone_count": len(milestones),
|
||||
"contributor_count": len(contribs),
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 渲染:Markdown 摘要报告
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def render_report(result: dict[str, Any]) -> str:
|
||||
repo = result["repo"]
|
||||
tl = result["timeline"]
|
||||
weeks = result["weeks"]
|
||||
meta = result["meta"]
|
||||
art = result["artifact_summary"]
|
||||
lines = [
|
||||
f"# 科研成果可视化沉淀报告 — {repo}\n",
|
||||
f"> 场景 S6 · 子赛题四「应用 GitLink 辅助科研」\n",
|
||||
f"## 一、活跃度概览(最近 {weeks} 周)\n",
|
||||
f"- 提交数: **{meta['commit_count']}**(窗口内峰值 "
|
||||
f"{max(tl['commits']) if tl['commits'] else 0} 提交/周)",
|
||||
f"- 新增 Issue: **{meta['issue_count']}**,新增 PR: **{meta['pr_count']}**",
|
||||
f"- 贡献者: **{meta['contributor_count']}**,里程碑: **{meta['milestone_count']}**\n",
|
||||
"## 二、开发节奏(最近 8 周快照)\n",
|
||||
"| 周 | commits | issues | prs |",
|
||||
"|----|---------|--------|-----|",
|
||||
]
|
||||
tail = tl["labels"][-8:]
|
||||
for i, label in enumerate(tail):
|
||||
idx = len(tl["labels"]) - len(tail) + i
|
||||
lines.append(f"| {label} | {tl['commits'][idx]} | {tl['issues'][idx]} | {tl['prs'][idx]} |")
|
||||
|
||||
lines += ["\n## 三、核心贡献者热力(贡献者 × 周提交数)\n",
|
||||
"| 贡献者 | 窗口内提交 |",
|
||||
"|--------|-----------|"]
|
||||
hm = result["heatmap"]
|
||||
for i, user in enumerate(hm["users"][:10]):
|
||||
total = sum(hm["matrix"][i])
|
||||
lines.append(f"| `{user}` | {total} |")
|
||||
|
||||
lines += ["\n## 四、科研产物分类\n",
|
||||
f"- 论文/笔记 (paper): **{art['paper']}**",
|
||||
f"- 数据集 (dataset): **{art['dataset']}**",
|
||||
f"- 模型 (model): **{art['model']}**",
|
||||
f"- 基准 (benchmark): **{art['benchmark']}**\n"]
|
||||
if result["paper_links"]:
|
||||
lines += ["## 五、抽取到的论文引用\n",
|
||||
"| 类型 | 链接 |",
|
||||
"|------|------|"]
|
||||
for p in result["paper_links"][:15]:
|
||||
lines.append(f"| {p['type']} | {p['target']} |")
|
||||
else:
|
||||
lines.append("## 五、抽取到的论文引用\n\n_未在 README/提交信息中发现 arXiv 或 DOI 引用_\n")
|
||||
|
||||
lines.append(f"\n_交互可视化见 visual.html(或原始数据 visual.json)_\n")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 渲染:交互 HTML(plotly 多子图)+ JSON bundle
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def render_html(result: dict[str, Any]) -> str | None:
|
||||
"""构建单一交互 HTML(多子图)。无 plotly 时返回 None(调用方降级)。"""
|
||||
try:
|
||||
import plotly.graph_objects as go
|
||||
from plotly.subplots import make_subplots
|
||||
except ImportError:
|
||||
return None
|
||||
|
||||
tl = result["timeline"]
|
||||
hm = result["heatmap"]
|
||||
langs = result["languages"] or {}
|
||||
|
||||
# 行布局:timeline(跨整行) | heatmap | pie | gantt
|
||||
fig = make_subplots(
|
||||
rows=4, cols=2,
|
||||
specs=[
|
||||
[{"colspan": 2}, None],
|
||||
[{"colspan": 2}, None],
|
||||
[{"type": "domain"}, {"type": "scatter"}],
|
||||
[{"colspan": 2}, None],
|
||||
],
|
||||
row_heights=[0.25, 0.30, 0.25, 0.20],
|
||||
vertical_spacing=0.08,
|
||||
subplot_titles=("开发时间线(周粒度)", "贡献者 × 周提交热力图",
|
||||
"语言占比", "里程碑甘特(due_date)"),
|
||||
)
|
||||
|
||||
# 1) 时间线折线
|
||||
fig.add_trace(go.Scatter(x=tl["labels"], y=tl["commits"], name="commits",
|
||||
mode="lines+markers", line=dict(color="#636efa")), row=1, col=1)
|
||||
fig.add_trace(go.Scatter(x=tl["labels"], y=tl["issues"], name="issues",
|
||||
mode="lines+markers", line=dict(color="#ef553b")), row=1, col=1)
|
||||
fig.add_trace(go.Scatter(x=tl["labels"], y=tl["prs"], name="prs",
|
||||
mode="lines+markers", line=dict(color="#00cc96")), row=1, col=1)
|
||||
|
||||
# 2) 贡献热力图
|
||||
if hm["users"]:
|
||||
fig.add_trace(go.Heatmap(
|
||||
z=hm["matrix"], x=hm["weeks"], y=hm["users"],
|
||||
colorscale="Blues", name="contributions",
|
||||
colorbar=dict(title="提交数", len=0.25, y=0.78),
|
||||
), row=2, col=1)
|
||||
|
||||
# 3a) 语言饼图
|
||||
if langs:
|
||||
labels = list(langs.keys())
|
||||
values = []
|
||||
for v in langs.values():
|
||||
# GitLink languages 形如 {"Python":"99.7%"};剥离 %
|
||||
s = str(v).strip().rstrip("%")
|
||||
try:
|
||||
values.append(float(s))
|
||||
except ValueError:
|
||||
values.append(0.0)
|
||||
fig.add_trace(go.Pie(labels=labels, values=values, name="languages",
|
||||
textinfo="label+percent"), row=3, col=1)
|
||||
|
||||
# 3b) 里程碑甘特(用散点的水平线段近似)
|
||||
gantt = result["milestones"]
|
||||
for g in gantt[:20]:
|
||||
title = g["title"] or "(milestone)"
|
||||
start = g["start"]
|
||||
due = g["due"]
|
||||
fig.add_trace(go.Scatter(
|
||||
x=[start, due], y=[title, title],
|
||||
mode="lines+markers",
|
||||
line=dict(color="#ffa15a", width=6),
|
||||
marker=dict(size=8),
|
||||
showlegend=False, hovertemplate=f"{title}<br>%{{x}}",
|
||||
), row=3, col=2)
|
||||
|
||||
fig.update_layout(
|
||||
title=f"科研成果可视化 — {result['repo']}(最近 {result['weeks']} 周)",
|
||||
height=1100, width=1100,
|
||||
legend=dict(orientation="h", y=1.02),
|
||||
margin=dict(l=40, r=40, t=80, b=40),
|
||||
)
|
||||
fig.update_xaxes(row=1, col=1, tickangle=-45)
|
||||
|
||||
return fig.to_html(full_html=True, include_plotlyjs="cdn",
|
||||
default_width="100%", default_height="1100px")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 主入口
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description="S6 科研成果可视化沉淀")
|
||||
ap.add_argument("--owner", required=True)
|
||||
ap.add_argument("--repo", required=True)
|
||||
ap.add_argument("--weeks", type=int, default=26, help="回溯多少周(默认 26)")
|
||||
ap.add_argument("--out", help="输出目录(写 visual.html + visual.json + report.md);"
|
||||
"省略则打印 JSON")
|
||||
args = ap.parse_args()
|
||||
|
||||
result = run(args.owner, args.repo, args.weeks)
|
||||
|
||||
if args.out:
|
||||
os.makedirs(args.out, exist_ok=True)
|
||||
# 原始数据 bundle(供前端二次开发)
|
||||
with open(os.path.join(args.out, "visual.json"), "w", encoding="utf-8") as f:
|
||||
json.dump(result, f, ensure_ascii=False, indent=2)
|
||||
with open(os.path.join(args.out, "report.md"), "w", encoding="utf-8") as f:
|
||||
f.write(render_report(result))
|
||||
|
||||
html = render_html(result)
|
||||
if html is not None:
|
||||
with open(os.path.join(args.out, "visual.html"), "w", encoding="utf-8") as f:
|
||||
f.write(html)
|
||||
print(f"✓ S6 可视化完成 → {args.out}/visual.html | visual.json | report.md")
|
||||
else:
|
||||
print(f"✓ S6 可视化完成(无 plotly,已降级)→ {args.out}/visual.json | report.md")
|
||||
print(" 提示:pip install plotly 后可生成交互 HTML")
|
||||
meta = result["meta"]
|
||||
art = result["artifact_summary"]
|
||||
print(f" commits={meta['commit_count']} issues={meta['issue_count']} "
|
||||
f"prs={meta['pr_count']} 贡献者={meta['contributor_count']}")
|
||||
print(f" 产物 paper={art['paper']} dataset={art['dataset']} "
|
||||
f"model={art['model']} benchmark={art['benchmark']}")
|
||||
print(f" 论文引用: {len(result['paper_links'])} 条")
|
||||
else:
|
||||
print(json.dumps(result, ensure_ascii=False, indent=2))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
@ -1,135 +0,0 @@
|
|||
package issue
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"reflect"
|
||||
"testing"
|
||||
|
||||
"github.com/gitlink-org/gitlink-cli/shortcuts/common"
|
||||
)
|
||||
|
||||
func TestParseCommaInts(t *testing.T) {
|
||||
got, err := parseCommaInts("1, 2,2, 3")
|
||||
if err != nil {
|
||||
t.Fatalf("parseCommaInts returned error: %v", err)
|
||||
}
|
||||
want := []int{1, 2, 3}
|
||||
if !reflect.DeepEqual(got, want) {
|
||||
t.Fatalf("parseCommaInts() = %#v, want %#v", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseCommaIntsRejectsInvalid(t *testing.T) {
|
||||
if _, err := parseCommaInts("1,abc"); err == nil {
|
||||
t.Fatal("parseCommaInts() expected error for non-integer")
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseCommaIntsEmpty(t *testing.T) {
|
||||
if _, err := parseCommaInts(""); err == nil {
|
||||
t.Fatal("parseCommaInts() expected error for empty input")
|
||||
}
|
||||
}
|
||||
|
||||
func TestBatchDeleteRequiresConfirm(t *testing.T) {
|
||||
server := common.NewTestServer(t, func(w http.ResponseWriter, r *http.Request) {
|
||||
t.Fatalf("no API calls expected without confirm, got %s %s", r.Method, r.URL.Path)
|
||||
})
|
||||
defer server.Close()
|
||||
|
||||
ctx := common.NewTestContext(t, server, "owner", "repo", map[string]string{
|
||||
"ids": "1,2",
|
||||
})
|
||||
err := common.RunShortcut(t, Shortcuts(), "batch-delete", ctx)
|
||||
if err != nil {
|
||||
t.Fatalf("batch-delete without confirm failed: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBatchDeleteWithConfirm(t *testing.T) {
|
||||
var deletePayload map[string]interface{}
|
||||
server := common.NewTestServer(t, func(w http.ResponseWriter, r *http.Request) {
|
||||
if r.Method == "DELETE" && r.URL.Path == "/v1/owner/repo/issues/batch_destroy.json" {
|
||||
deletePayload = common.DecodeJSON(t, r)
|
||||
common.WriteJSON(t, w, map[string]interface{}{
|
||||
"status": 0,
|
||||
"message": "success",
|
||||
})
|
||||
} else {
|
||||
t.Fatalf("unexpected request: %s %s", r.Method, r.URL.Path)
|
||||
}
|
||||
})
|
||||
defer server.Close()
|
||||
|
||||
ctx := common.NewTestContext(t, server, "owner", "repo", map[string]string{
|
||||
"ids": "10,20,30",
|
||||
"confirm": "true",
|
||||
})
|
||||
err := common.RunShortcut(t, Shortcuts(), "batch-delete", ctx)
|
||||
if err != nil {
|
||||
t.Fatalf("batch-delete with confirm failed: %v", err)
|
||||
}
|
||||
|
||||
ids, ok := deletePayload["ids"].([]interface{})
|
||||
if !ok {
|
||||
t.Fatalf("ids not a slice: %T", deletePayload["ids"])
|
||||
}
|
||||
if len(ids) != 3 {
|
||||
t.Fatalf("expected 3 ids, got %d", len(ids))
|
||||
}
|
||||
}
|
||||
|
||||
func TestBatchUpdateDryRun(t *testing.T) {
|
||||
server := common.NewTestServer(t, func(w http.ResponseWriter, r *http.Request) {
|
||||
t.Fatalf("no API calls expected in dry-run, got %s %s", r.Method, r.URL.Path)
|
||||
})
|
||||
defer server.Close()
|
||||
|
||||
ctx := common.NewTestContext(t, server, "owner", "repo", map[string]string{
|
||||
"ids": "1,2,3",
|
||||
"status": "closed",
|
||||
"dry-run": "true",
|
||||
})
|
||||
err := common.RunShortcut(t, Shortcuts(), "batch-update", ctx)
|
||||
if err != nil {
|
||||
t.Fatalf("batch-update dry-run failed: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBatchUpdateApply(t *testing.T) {
|
||||
var updatePayload map[string]interface{}
|
||||
server := common.NewTestServer(t, func(w http.ResponseWriter, r *http.Request) {
|
||||
if r.Method == "PATCH" && r.URL.Path == "/v1/owner/repo/issues/batch_update.json" {
|
||||
updatePayload = common.DecodeJSON(t, r)
|
||||
common.WriteJSON(t, w, map[string]interface{}{
|
||||
"status": 0,
|
||||
"message": "success",
|
||||
})
|
||||
} else {
|
||||
t.Fatalf("unexpected request: %s %s", r.Method, r.URL.Path)
|
||||
}
|
||||
})
|
||||
defer server.Close()
|
||||
|
||||
ctx := common.NewTestContext(t, server, "owner", "repo", map[string]string{
|
||||
"ids": "10,20",
|
||||
"status": "closed",
|
||||
"milestone": "5",
|
||||
"dry-run": "false",
|
||||
})
|
||||
err := common.RunShortcut(t, Shortcuts(), "batch-update", ctx)
|
||||
if err != nil {
|
||||
t.Fatalf("batch-update failed: %v", err)
|
||||
}
|
||||
|
||||
ids, ok := updatePayload["ids"].([]interface{})
|
||||
if !ok {
|
||||
t.Fatalf("ids not a slice: %T", updatePayload["ids"])
|
||||
}
|
||||
if len(ids) != 2 {
|
||||
t.Fatalf("expected 2 ids, got %d", len(ids))
|
||||
}
|
||||
if updatePayload["milestone_id"] != float64(5) {
|
||||
t.Fatalf("expected milestone_id=5, got %v", updatePayload["milestone_id"])
|
||||
}
|
||||
}
|
||||
|
|
@ -114,7 +114,7 @@ func Shortcuts() []*common.Shortcut {
|
|||
body := map[string]interface{}{
|
||||
"status": "closed",
|
||||
}
|
||||
env, err := ctx.CallAPI("POST", fmt.Sprintf("%s/%s/milestones/%s/update_status", ctx.RepoPath(), ctx.Owner, id), body)
|
||||
env, err := ctx.CallAPI("POST", fmt.Sprintf("%s/milestones/%s/update_status", v1Path(ctx), id), body)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
|
|
|||
|
|
@ -115,3 +115,30 @@ func TestMilestoneDelete(t *testing.T) {
|
|||
t.Fatalf("delete failed: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMilestoneClose(t *testing.T) {
|
||||
var closePayload map[string]interface{}
|
||||
server := common.NewTestServer(t, func(w http.ResponseWriter, r *http.Request) {
|
||||
// Regression guard: +close must hit /v1/{owner}/{repo}/milestones/{id}/update_status
|
||||
// (previously malformed to /{owner}/{repo}/{owner}/milestones/{id}/update_status — Owner duplicated, Repo dropped, no /v1).
|
||||
if r.Method == "POST" && r.URL.Path == "/v1/owner/repo/milestones/1/update_status.json" {
|
||||
closePayload = common.DecodeJSON(t, r)
|
||||
common.WriteJSON(t, w, map[string]interface{}{
|
||||
"status": 0,
|
||||
"message": "更新成功",
|
||||
})
|
||||
} else {
|
||||
t.Fatalf("unexpected request: %s %s", r.Method, r.URL.Path)
|
||||
}
|
||||
})
|
||||
defer server.Close()
|
||||
|
||||
ctx := common.NewTestContext(t, server, "owner", "repo", map[string]string{
|
||||
"id": "1",
|
||||
})
|
||||
err := common.RunShortcut(t, Shortcuts(), "close", ctx)
|
||||
if err != nil {
|
||||
t.Fatalf("close failed: %v", err)
|
||||
}
|
||||
common.AssertEqual(t, closePayload["status"], "closed")
|
||||
}
|
||||
|
|
|
|||
|
|
@ -13,10 +13,8 @@ func TestRegisterAll(t *testing.T) {
|
|||
expectedGroups := []string{
|
||||
"repo", "issue", "label", "license", "pr", "profile", "release", "branch",
|
||||
"org", "user", "search", "ci", "workflow",
|
||||
"compare", "member", "milestone", "pipeline", "pm", "webhook",
|
||||
"dataset", "health", "ignore",
|
||||
// HEAD-only groups preserved during the upstream/master merge.
|
||||
"file", "snippet", "wiki",
|
||||
"compare", "member", "milestone", "pipeline", "webhook",
|
||||
"dataset", "health", "ignore", "file", "snippet", "pm", "wiki",
|
||||
}
|
||||
|
||||
groupSet := map[string]bool{}
|
||||
|
|
|
|||
|
|
@ -170,6 +170,25 @@ skills/
|
|||
|
||||
> 以上 5 个为本次新增的 AI 工作流 Skill,均兼容 Claude Code 等 Agent,详见各 `SKILL.md`。
|
||||
|
||||
### 科研辅助 Skills(子赛题四「应用 GitLink 辅助科研」)
|
||||
|
||||
采用「Go 出数据 + Python 做算法」:数据复用现有 gitlink-cli 域,科研算法在 `scripts/research/*.py`(networkx/plotly),每个场景配可复现脚本与 Skill 规范,并可通过 `gitlink-cli server` 网页终端演示。详见 [../doc/科研场景使用指南.md](../doc/科研场景使用指南.md)。
|
||||
|
||||
| Skill | 场景 | 说明 | 命令 |
|
||||
|-------|------|------|------|
|
||||
| **gitlink-research-insight** | S1 | 仓库级科研项目洞悉:演进谱系 + 创新点 | `python scripts/research/lineage.py` |
|
||||
| **gitlink-research-graph** | S2 | 科研知识图谱(networkx 节点/边)+ 热点追踪 | `python scripts/research/graph_build.py` |
|
||||
| **gitlink-compliance** | S3 | 合规与复现性检查(license/密钥/复现) | `python scripts/research/repro.py` |
|
||||
| **gitlink-collab-match** | S4 | 科研协作智能匹配(缺口×画像) | `python scripts/research/match.py` |
|
||||
| **gitlink-research-progress** | S5 | 进度智能跟踪与预警(周报+风险) | `python scripts/research/report.py` |
|
||||
| **gitlink-research-visual** | S6 | 科研成果可视化(plotly 交互图表) | `python scripts/research/visual.py` |
|
||||
| gitlink-research-tracker | S2/S5 | 技术调研与热点追踪(含真机 Agent 日志) | 见 SKILL.md |
|
||||
| gitlink-license-compliance | S3 | 许可证深度合规扫描 | 见 SKILL.md |
|
||||
| gitlink-scholar-profile | S4/S6 | 学者/团队科研画像 | 见 SKILL.md |
|
||||
| gitlink-research-fork-impact | S1/S6 | Fork 影响力与想法传播分析 | 见 SKILL.md |
|
||||
|
||||
> 6 个场景均已在真实科研仓库 `mindspore-Ecosystem/mindspore` 上验证;技术实现详见 [../doc/科研场景技术实现报告.md](../doc/科研场景技术实现报告.md)。
|
||||
|
||||
---
|
||||
|
||||
## 🎯 使用场景
|
||||
|
|
|
|||
|
|
@ -0,0 +1,76 @@
|
|||
---
|
||||
name: gitlink-collab-match
|
||||
version: 1.0.0
|
||||
description: "科研协作智能匹配(子赛题四·S4):分析科研仓库的技术缺口(未解决 Issue 主题/语言、开放 PR、研究空缺),结合候选人科研画像,智能匹配跨团队/跨学者协作伙伴。当用户要找协作者、推荐合作者、分析仓库需要什么样的人时触发。"
|
||||
metadata:
|
||||
requires:
|
||||
bins: ["gitlink-cli"]
|
||||
python: ["scripts/research/requirements.txt"]
|
||||
cliHelp: "gitlink-cli research +match --help"
|
||||
scenario: "S4"
|
||||
---
|
||||
|
||||
# gitlink-collab-match — 科研协作智能匹配
|
||||
|
||||
> 子赛题四「应用 GitLink 辅助科研」· 场景 **S4 科研协作智能匹配**
|
||||
|
||||
## 何时使用
|
||||
|
||||
- 课题组/科研团队想为一个科研代码仓库寻找合适的协作伙伴(跨团队/跨学者)。
|
||||
- 想知道「这个仓库当前最缺哪方面的人/技能」。
|
||||
- 为开源科研项目做人员招募建议、互补团队推荐。
|
||||
|
||||
## 前置条件
|
||||
|
||||
1. 已 `gitlink-cli auth login`(Token 7 天有效)。
|
||||
2. 已 `pip install -r scripts/research/requirements.txt`(本场景实际只用标准库 + topics 词典,无需重型依赖)。
|
||||
3. 目标仓库存在且有若干未解决 Issue(缺口信号来源)。
|
||||
|
||||
## 工作流
|
||||
|
||||
本 Skill 的算法由 `scripts/research/match.py` 实现(Go 出数据 + Python 做匹配):
|
||||
|
||||
1. **缺口分析**:调 `issue +list --state open`(按优先级加权)+ `pr +list --state open` + `repo +languages` + README,用 `topics.py` 词典抽取出仓库的**缺口主题向量**与**需求语言**。
|
||||
2. **候选池**:本仓库贡献者(`repo +contributors`,过滤 bot)+ 按缺口主题用 `search +users` 搜到的外部用户,上限默认 15。
|
||||
3. **候选人画像**:对每个候选人调 `repo +list --user <login>`,聚合其公开仓库的主题向量、语言集合、fork 数(协作开放度)、活跃度。
|
||||
4. **综合打分**:
|
||||
`score = 0.45×主题重叠(余弦) + 0.20×语言匹配(Jaccard) + 0.20×活跃度 + 0.15×协作开放度`(×100)。
|
||||
5. **产物**:`match.json`(结构化)+ `report.md`(中文推荐报告,含缺口表 + 排名表 + 理由)+ `network.mmd`(Mermaid 协作网络图)。
|
||||
|
||||
## 命令
|
||||
|
||||
```bash
|
||||
# 默认输出到 stdout(JSON)
|
||||
python scripts/research/match.py --owner mindspore-Ecosystem --repo mindspore
|
||||
|
||||
# 输出三件产物到目录
|
||||
python scripts/research/match.py --owner <OWNER> --repo <REPO> --top 10 --pool 15 --out ./out
|
||||
|
||||
# 可复现脚本(封装了上述流程)
|
||||
bash skills/gitlink-collab-match/examples/collab-match-workflow.sh <OWNER> <REPO> [OUT_DIR]
|
||||
```
|
||||
|
||||
## 输出结构(match.json)
|
||||
|
||||
```json
|
||||
{
|
||||
"scenario": "S4_collaboration_matching",
|
||||
"repo": "owner/repo",
|
||||
"gap_topics": ["deep_learning", "computer_vision", "..."],
|
||||
"needed_languages": ["python", "..."],
|
||||
"gap_signals": [{"type":"unresolved_issue","topic":"deep_learning","evidence":"...","priority":"高"}],
|
||||
"candidates": [{"login":"...","score":17.0,"topic_overlap":0.31,"language_match":0.5,
|
||||
"activity_level":"high","repo_languages":["python"],"reasons":["覆盖缺口主题: ..."]}]
|
||||
}
|
||||
```
|
||||
|
||||
## 验证
|
||||
|
||||
已在真实科研仓库 **`mindspore-Ecosystem/mindspore`**(20346 条 issue)上验证:
|
||||
缺口主题正确识别为 deep_learning / scientific_computing / RL / CV 等;
|
||||
Top 推荐为仓库真实活跃贡献者(yefeng / He_Wei / gaoyong10)。
|
||||
|
||||
## 兼容性
|
||||
|
||||
兼容 Claude Code 等 AI Agent:本 SKILL.md 即为 Agent 编排依据,
|
||||
Agent 可直接调上述命令并把产物读回做进一步解读与文案化。
|
||||
|
|
@ -0,0 +1,45 @@
|
|||
#!/usr/bin/env bash
|
||||
# collab-match-workflow.sh — S4 科研协作智能匹配 · 可复现执行脚本
|
||||
# 子赛题四「应用 GitLink 辅助科研」交付物之一
|
||||
#
|
||||
# 用法: bash collab-match-workflow.sh <OWNER> <REPO> [OUT_DIR] [POOL] [TOP] [ISSUE_SAMPLE]
|
||||
# 示例: bash collab-match-workflow.sh mindspore-Ecosystem mindspore ./out 15 10 100
|
||||
set -euo pipefail
|
||||
|
||||
OWNER="${1:?用法: $0 <OWNER> <REPO> [OUT_DIR] [POOL] [TOP] [ISSUE_SAMPLE]}"
|
||||
REPO="${2:?缺少 REPO}"
|
||||
OUT_DIR="${3:-./collab-match-output}"
|
||||
POOL="${4:-15}"
|
||||
TOP="${5:-10}"
|
||||
ISSUE_SAMPLE="${6:-100}"
|
||||
|
||||
# 定位仓库根(脚本位于 skills/gitlink-collab-match/examples/)
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
REPO_ROOT="$(cd "$SCRIPT_DIR/../../.." && pwd)"
|
||||
MATCH="$REPO_ROOT/scripts/research/match.py"
|
||||
|
||||
# 本地 Windows 开发可设 GITLINK_CLI=./gitlink-cli.exe;Linux/容器走 PATH 默认值
|
||||
: "${GITLINK_CLI:=gitlink-cli}"
|
||||
|
||||
echo "==> 目标仓库: $OWNER/$REPO"
|
||||
echo "==> CLI: $GITLINK_CLI"
|
||||
echo "==> 候选池=$POOL Top=$TOP Issue采样=$ISSUE_SAMPLE"
|
||||
|
||||
mkdir -p "$OUT_DIR"
|
||||
GITLINK_CLI="$GITLINK_CLI" python "$MATCH" \
|
||||
--owner "$OWNER" --repo "$REPO" \
|
||||
--pool "$POOL" --top "$TOP" --issue-sample "$ISSUE_SAMPLE" \
|
||||
--out "$OUT_DIR"
|
||||
|
||||
echo
|
||||
echo "==> 产物:"
|
||||
ls -1 "$OUT_DIR"
|
||||
echo
|
||||
echo "==> Top 推荐预览:"
|
||||
python -c "
|
||||
import json,sys
|
||||
d=json.load(open('$OUT_DIR/match.json',encoding='utf-8'))
|
||||
print('缺口主题:', ', '.join(d['gap_topics']))
|
||||
for i,m in enumerate(d['candidates'],1):
|
||||
print(f\" {i}. {m['login']} ({m['score']}分) — {'; '.join(m['reasons'][:2])}\")
|
||||
"
|
||||
|
|
@ -6,20 +6,38 @@ metadata:
|
|||
requires:
|
||||
bins: ["gitlink-cli"]
|
||||
cliHelp: "gitlink-cli repo --help"
|
||||
scenario: "S3"
|
||||
---
|
||||
|
||||
# gitlink-compliance(开源合规检查)
|
||||
# gitlink-compliance(开源合规与复现性检查)
|
||||
|
||||
**CRITICAL — 开始前必须先阅读 [`../gitlink-shared/SKILL.md`](../gitlink-shared/SKILL.md),其中包含认证、权限处理和 API 注意事项。**
|
||||
**CRITICAL — GitLink 操作只能用 `gitlink-cli`。禁止用 `gh`(GitHub CLI)操作 GitLink 资源。`gh` 仅适用于 GitHub 平台。**
|
||||
|
||||
> **前置条件:** 先阅读 [`../gitlink-shared/SKILL.md`](../gitlink-shared/SKILL.md) 了解认证和全局参数。
|
||||
|
||||
子赛题四「应用 GitLink 辅助科研」· 场景 **S3 科研项目合规与复现性检查** 的自动化算法由
|
||||
`scripts/research/repro.py`(Go 出数据 + Python 做算法)实现,见 **工作流 4**。
|
||||
|
||||
---
|
||||
|
||||
## 何时使用
|
||||
|
||||
- 科研项目准备开源发布前,做一次全面的合规与复现性审查。
|
||||
- 想知道「这个仓库别人能不能复现」:CI、lockfile、README 复现说明、版本 tag、容器环境是否齐备。
|
||||
- 想发现仓库里的合规风险与敏感信息泄露:缺 LICENSE / 版权头、数据目录入库、`.env` 泄露、硬编码密钥。
|
||||
|
||||
## 前置条件
|
||||
|
||||
1. 已 `gitlink-cli auth login`(Token 7 天有效)。
|
||||
2. 目标仓库存在且可读取文件树(`repo +info` / `file +get` / `repo +tree`)。
|
||||
3. 复现性自动化检查(工作流 4)只用 Python 标准库,无需第三方依赖。
|
||||
|
||||
---
|
||||
|
||||
## 工作流概览
|
||||
|
||||
本 Skill 提供开源项目的合规性自动化检查能力,帮助 Maintainer 在发布前发现并修复合规问题。
|
||||
本 Skill 提供开源项目的合规性与复现性自动化检查能力,帮助 Maintainer 在发布前发现并修复合规问题。
|
||||
|
||||
| 检查类型 | 覆盖范围 | 严重程度 |
|
||||
|----------|---------|:--------:|
|
||||
|
|
@ -28,6 +46,8 @@ metadata:
|
|||
| 依赖合规 | 第三方依赖许可证兼容性 | 🔴 |
|
||||
| 安全策略 | SECURITY.md、安全披露流程 | 🟡 |
|
||||
| 贡献者协议 | CLA / DCO 要求 | 🔵 |
|
||||
| 复现性 | CI / lockfile / README 复现说明 / 版本 tag / 容器环境 | 🟡 |
|
||||
| 数据隐私 | data/ 入库、.env 泄露、密钥硬编码 | 🔴 |
|
||||
|
||||
---
|
||||
|
||||
|
|
@ -206,6 +226,78 @@ gitlink-cli api GET /:owner/:repo/raw/master/src/main.py
|
|||
|
||||
---
|
||||
|
||||
## 工作流 4:合规与复现性自动化检查(repro.py)
|
||||
|
||||
**场景**:子赛题四·S3 科研项目合规与复现性检查 —— 对一个科研仓库同时给出「合规分」与「复现分」,并产出检查清单、风险项与中文报告。
|
||||
|
||||
本工作流的算法由 `scripts/research/repro.py` 实现,数据全部经 gitlink-cli 获取(Go 出数据 + Python 做算法)。
|
||||
|
||||
### 数据采集
|
||||
|
||||
`repro.py` 内部调用以下 gitlink-cli 命令(已封装在 `collect.py` 中):
|
||||
|
||||
```bash
|
||||
# 仓库信息(默认分支、版本 tag)
|
||||
gitlink-cli --owner <owner> --repo <repo> repo +info --format json
|
||||
|
||||
# 关键文件文本(LICENSE / README / go.mod / requirements.txt / package.json / .gitignore / SECURITY.md / ...)
|
||||
gitlink-cli --owner <owner> --repo <repo> file +get --path LICENSE --ref master
|
||||
|
||||
# 根文件树(扫 data/、.env、config、.gitea/.github workflows 等是否存在)
|
||||
gitlink-cli --owner <owner> --repo <repo> repo +tree --ref master
|
||||
|
||||
# 语言占比(仅作为元信息记录)
|
||||
gitlink-cli --owner <owner> --repo <repo> repo +languages --format json
|
||||
```
|
||||
|
||||
### 算法(纯函数,可单测)
|
||||
|
||||
| 函数 | 作用 |
|
||||
|------|------|
|
||||
| `identify_license(text)` | 关键词匹配 MulanPSL / Apache / MIT / GPL / LGPL / BSD / ISC / MPL / 无 |
|
||||
| `scan_secrets(text, file)` | 正则找 private key / AWS token / API key / Slack / GitHub token / JWT / 邮箱 / 手机号 → `[{level,category,file,line,detail}]`(脱敏) |
|
||||
| `repro_checks(file_texts, tree, repo_info)` | CI 配置、lockfile、README 复现说明、版本 tag、容器化,每项 `{name,pass,score(0-2),evidence}` |
|
||||
| `compliance_items(license_info, file_texts, tree)` | LICENSE 声明、SECURITY.md、版权头、依赖合规、CONTRIBUTING.md |
|
||||
| `data_privacy(tree, gitignore_text)` | data/ 入库、.env 入库、.gitignore 是否忽略 .env |
|
||||
|
||||
打分:`repro_score` / `compliance_score` 均为 0-10(各项 0-2 分聚合归一)。
|
||||
|
||||
### 命令
|
||||
|
||||
```bash
|
||||
# 默认输出到 stdout(JSON)
|
||||
python scripts/research/repro.py --owner mindspore-Ecosystem --repo mindspore
|
||||
|
||||
# 输出两件产物到目录(repro.json + compliance_report.md)
|
||||
python scripts/research/repro.py --owner <OWNER> --repo <REPO> --out ./out
|
||||
|
||||
# 可复现脚本(封装了上述流程)
|
||||
bash skills/gitlink-compliance/examples/compliance-repro-workflow.sh <OWNER> <REPO> [OUT_DIR]
|
||||
```
|
||||
|
||||
### 输出结构(repro.json)
|
||||
|
||||
```json
|
||||
{
|
||||
"scenario": "S3_compliance_reproducibility",
|
||||
"repo": "owner/repo",
|
||||
"default_branch": "master",
|
||||
"license": "MIT",
|
||||
"repro_items": [{"name": "CI 配置", "pass": true, "score": 2, "evidence": "..."}],
|
||||
"compliance_items": [{"name": "LICENSE 文件", "pass": true, "score": 2, "evidence": "..."}],
|
||||
"privacy_items": [{"name": ".env 入库", "pass": true, "score": 2, "evidence": "..."}],
|
||||
"secrets": [{"level": "critical", "category": "private_key", "file": "config.env", "line": 5, "detail": "..."}],
|
||||
"risks": [{"area": "secret", "name": "private_key", "file": "...", "level": "critical", "evidence": "..."}],
|
||||
"repro_score": 8.0,
|
||||
"compliance_score": 6.0,
|
||||
"meta": {"key_files_found": ["LICENSE", "README.md"], "tree_size": 42, "languages": {"Python": "99%"}}
|
||||
}
|
||||
```
|
||||
|
||||
`compliance_report.md` 包含:复现性检查清单表、合规性检查清单表、数据隐私检查表、风险项表(按严重程度排序)与打分。
|
||||
|
||||
---
|
||||
|
||||
## Raw API 参考
|
||||
|
||||
```bash
|
||||
|
|
|
|||
|
|
@ -0,0 +1,45 @@
|
|||
#!/usr/bin/env bash
|
||||
# compliance-repro-workflow.sh — S3 科研项目合规与复现性检查 · 可复现执行脚本
|
||||
# 子赛题四「应用 GitLink 辅助科研」交付物之一
|
||||
#
|
||||
# 用法: bash compliance-repro-workflow.sh <OWNER> <REPO> [OUT_DIR]
|
||||
# 示例: bash compliance-repro-workflow.sh mindspore-Ecosystem mindspore ./out
|
||||
set -euo pipefail
|
||||
|
||||
OWNER="${1:?用法: $0 <OWNER> <REPO> [OUT_DIR]}"
|
||||
REPO="${2:?缺少 REPO}"
|
||||
OUT_DIR="${3:-./compliance-repro-output}"
|
||||
|
||||
# 定位仓库根(脚本位于 skills/gitlink-compliance/examples/)
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
REPO_ROOT="$(cd "$SCRIPT_DIR/../../.." && pwd)"
|
||||
REPRO="$REPO_ROOT/scripts/research/repro.py"
|
||||
|
||||
# 本地 Windows 开发可设 GITLINK_CLI=./gitlink-cli.exe;Linux/容器走 PATH 默认值
|
||||
: "${GITLINK_CLI:=gitlink-cli}"
|
||||
|
||||
echo "==> 目标仓库: $OWNER/$REPO"
|
||||
echo "==> CLI: $GITLINK_CLI"
|
||||
echo "==> 输出目录: $OUT_DIR"
|
||||
|
||||
mkdir -p "$OUT_DIR"
|
||||
GITLINK_CLI="$GITLINK_CLI" python "$REPRO" \
|
||||
--owner "$OWNER" --repo "$REPO" \
|
||||
--out "$OUT_DIR"
|
||||
|
||||
echo
|
||||
echo "==> 产物:"
|
||||
ls -1 "$OUT_DIR"
|
||||
echo
|
||||
echo "==> 合规/复现检查摘要:"
|
||||
python -c "
|
||||
import json
|
||||
d=json.load(open('$OUT_DIR/repro.json',encoding='utf-8'))
|
||||
print('许可证:', d['license'])
|
||||
print('复现分: %s/10' % d['repro_score'])
|
||||
print('合规分: %s/10' % d['compliance_score'])
|
||||
print('风险项: %d 处' % len(d['risks']))
|
||||
for r in d['risks'][:5]:
|
||||
loc = (r.get('file','') + ':' + str(r.get('line',''))) if r.get('file') else '-'
|
||||
print(' [%s] %s (%s) %s' % (r.get('level','medium'), r.get('name',''), loc, r.get('evidence') or r.get('detail','')))
|
||||
"
|
||||
|
|
@ -0,0 +1,87 @@
|
|||
---
|
||||
name: gitlink-research-graph
|
||||
version: 1.0.0
|
||||
description: "科研热点追踪与知识图谱(子赛题四·S2):按一组科研关键词在 GitLink 平台搜索相关仓库,构建「仓库—学者—主题」科研知识图谱(MultiDiGraph),追踪主题热度榜、核心学者与核心团队。当用户要梳理某个研究方向的全景、画知识图谱、找热点/核心学者/团队时触发。"
|
||||
metadata:
|
||||
requires:
|
||||
bins: ["gitlink-cli"]
|
||||
python: ["scripts/research/requirements.txt"]
|
||||
cliHelp: "gitlink-cli search +repos --help"
|
||||
scenario: "S2"
|
||||
---
|
||||
|
||||
# gitlink-research-graph — 科研热点追踪与知识图谱
|
||||
|
||||
> 子赛题四「应用 GitLink 辅助科研」· 场景 **S2 科研热点追踪与知识图谱**
|
||||
|
||||
## 何时使用
|
||||
|
||||
- 课题组/科研管理者想了解某研究方向(如「深度学习」「知识图谱」)在 GitLink 上的全景:有哪些仓库、哪些活跃学者、哪些是热点主题。
|
||||
- 想把「关键词 → 仓库 → 学者/主题」的关系可视化为一张知识图谱(Mermaid / Graphviz)。
|
||||
- 为开题、综述、找人合作做主题态势感知。
|
||||
|
||||
## 前置条件
|
||||
|
||||
1. 已 `gitlink-cli auth login`(Token 7 天有效)。
|
||||
2. 已 `pip install -r scripts/research/requirements.txt`(本场景需 **networkx** 做图谱构建)。
|
||||
3. 准备好一组逗号分隔的关键词(如 `"deep learning,nlp,knowledge graph"`)。
|
||||
|
||||
## 工作流
|
||||
|
||||
本 Skill 的算法由 `scripts/research/graph_build.py` 实现(Go 出数据 + Python 做建图)。严格分离「取数」与「建图」,便于离线单测:
|
||||
|
||||
1. **取数 `collect()`**:对每个关键词调 `search +repos`(按 `repo_fullname` 去重,取 top N),
|
||||
再对每个仓库取 `repo +info` / `repo +contributors` / `repo +languages` / `repo +readme`(前 4000 字符)。
|
||||
2. **建图 `build_graph(repos, contributors_map, languages_map, readmes)`**(纯函数,不联网):
|
||||
用 `networkx.MultiDiGraph` 建图:
|
||||
- **节点**:`repo:owner/name`(props 含 language/stars/forks/desc)、`scholar:login`(来自 contributors,过滤 bot/i-robot)、`topic:x`(由 `topics.py` 词典在 description+readme 上抽取)。
|
||||
- **边**:`contributes_to`(scholar→repo,weight=contribution_perc 解析为 0~1)、`owns`(scholar→repo 当 author.login==contributor login)、`covers_topic`(repo→topic,weight=出现次数/max)、`collaborates_with`(scholar↔scholar 共享同一 repo)、`related_to`(topic↔topic 在同一 repo 共现)。
|
||||
3. **产物**:`graph.json`(结构化,含 scenario/nodes/edges/core_scholars/topic_heat/meta)+ `report.md`(中文趋势报告)+ `graph.mmd`(Mermaid,按 repo/scholar/topic 着色,仅画前 ~40 节点防爆炸)+ `graph.dot`(Graphviz DOT)。
|
||||
4. **趋势报告**:主题热度榜(`topics.topic_counter` 在全部 description 上的 top10)+ 核心学者(按出现 repo 数)+ 核心团队。
|
||||
|
||||
## 命令
|
||||
|
||||
```bash
|
||||
# 默认输出到 stdout(JSON)
|
||||
python scripts/research/graph_build.py --keywords "deep learning,nlp"
|
||||
|
||||
# 输出四件产物到目录
|
||||
python scripts/research/graph_build.py --keywords "knowledge graph,gnn" --repos-limit 20 --out ./out
|
||||
|
||||
# 可复现脚本(封装了上述流程)
|
||||
bash skills/gitlink-research-graph/examples/knowledge-graph-workflow.sh <OWNER_KEYWORDS_EXAMPLE>
|
||||
# 实际签名:bash knowledge-graph-workflow.sh <KEYWORDS> [OUT_DIR] [REPOS_LIMIT]
|
||||
```
|
||||
|
||||
## 输出结构(graph.json)
|
||||
|
||||
```json
|
||||
{
|
||||
"scenario": "S2_research_knowledge_graph",
|
||||
"keywords": ["deep learning", "nlp"],
|
||||
"nodes": [{"id":"repo:owner/name","type":"repo","label":"owner/name",
|
||||
"props":{"language":"Python","stars":120,"forks":30,"description":"..."}}],
|
||||
"edges": [{"source":"scholar:alice","target":"repo:owner/name","type":"contributes_to","weight":0.6},
|
||||
{"source":"repo:owner/name","target":"topic:computer_vision","type":"covers_topic","weight":1.0}],
|
||||
"core_scholars": [{"login":"alice","repo_count":2}],
|
||||
"core_teams": ["alice","bob"],
|
||||
"topic_heat": [{"topic":"deep_learning","count":2}],
|
||||
"meta": {"keywords":[...], "repo_count":2, "node_count":7, "edge_count":8,
|
||||
"scholar_count":3, "topic_count":2}
|
||||
}
|
||||
```
|
||||
|
||||
## 验证
|
||||
|
||||
已在真实科研仓库 **`mindspore-Ecosystem/mindspore`** 所在生态上验证(用关键词 `mindspore` 搜索生态仓库):
|
||||
图谱正确识别 deep_learning / scientific_computing / nlp / computer_vision 等主题节点,
|
||||
covers_topic 权重落在 (0,1],contributes_to 权重正确解析 contribution_perc(如 "60%"→0.6),
|
||||
bot(i-robot)账号被过滤。Mermaid 输出以 `graph TD` 开头、含 repo/scholar/topic 三色 classDef,节点数被截断到 40 以内。
|
||||
|
||||
单测:`python scripts/research/test_graph_build.py`(17 个用例,不联网,构造 mock 数据喂 `build_graph`)。
|
||||
|
||||
## 兼容性
|
||||
|
||||
兼容 Claude Code 等 AI Agent:本 SKILL.md 即为 Agent 编排依据,
|
||||
Agent 可直接调上述命令并把 graph.json/graph.mmd 读回做进一步解读与文案化。
|
||||
Mermaid 块可被支持 Mermaid 渲染的 Markdown 查看器直接展示;graph.dot 可用 `dot -Tsvg graph.dot -o graph.svg` 渲染。
|
||||
|
|
@ -0,0 +1,43 @@
|
|||
#!/usr/bin/env bash
|
||||
# knowledge-graph-workflow.sh — S2 科研热点追踪与知识图谱 · 可复现执行脚本
|
||||
# 子赛题四「应用 GitLink 辅助科研」交付物之一
|
||||
#
|
||||
# 用法: bash knowledge-graph-workflow.sh <KEYWORDS> [OUT_DIR] [REPOS_LIMIT]
|
||||
# 示例: bash knowledge-graph-workflow.sh "deep learning,nlp" ./out 20
|
||||
set -euo pipefail
|
||||
|
||||
KEYWORDS="${1:?用法: $0 <KEYWORDS> [OUT_DIR] [REPOS_LIMIT] 例如: $0 \"deep learning,nlp\" ./out 20}"
|
||||
OUT_DIR="${2:-./knowledge-graph-output}"
|
||||
REPOS_LIMIT="${3:-20}"
|
||||
|
||||
# 定位仓库根(脚本位于 skills/gitlink-research-graph/examples/)
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
REPO_ROOT="$(cd "$SCRIPT_DIR/../../.." && pwd)"
|
||||
GRAPH="$REPO_ROOT/scripts/research/graph_build.py"
|
||||
|
||||
# 本地 Windows 开发可设 GITLINK_CLI=./gitlink-cli.exe;Linux/容器走 PATH 默认值
|
||||
: "${GITLINK_CLI:=gitlink-cli}"
|
||||
|
||||
echo "==> 关键词: $KEYWORDS"
|
||||
echo "==> CLI: $GITLINK_CLI"
|
||||
echo "==> 仓库上限: $REPOS_LIMIT"
|
||||
|
||||
mkdir -p "$OUT_DIR"
|
||||
GITLINK_CLI="$GITLINK_CLI" python "$GRAPH" \
|
||||
--keywords "$KEYWORDS" \
|
||||
--repos-limit "$REPOS_LIMIT" \
|
||||
--out "$OUT_DIR"
|
||||
|
||||
echo
|
||||
echo "==> 产物:"
|
||||
ls -1 "$OUT_DIR"
|
||||
echo
|
||||
echo "==> 图谱概览:"
|
||||
python -c "
|
||||
import json
|
||||
d=json.load(open('$OUT_DIR/graph.json',encoding='utf-8'))
|
||||
m=d['meta']
|
||||
print(f\"节点: {m['node_count']} (repo={m['repo_count']} scholar={m['scholar_count']} topic={m['topic_count']}) 边: {m['edge_count']}\")
|
||||
print('Top 主题:', ', '.join(h['topic'] for h in d['topic_heat'][:5]))
|
||||
print('核心学者:', ', '.join(f\"{s['login']}({s['repo_count']})\" for s in d['core_scholars'][:5]))
|
||||
"
|
||||
|
|
@ -1,11 +1,13 @@
|
|||
---
|
||||
name: gitlink-research-insight
|
||||
version: 1.0.0
|
||||
description: "科研仓库画像:采集 GitLink 科研项目数据,从可复现性、活跃度、引用价值、协作健康四个维度评估,输出科研评估报告与协作知识图谱。当用户提到「科研分析」「科研项目评估」「仓库画像」「可复现性」「引用价值」「科研洞悉」「research insight」时触发。"
|
||||
description: "科研仓库洞悉:场景 S1 仓库级科研项目谱系(lineage)分析 + 四维科研画像(可复现性/活跃度/引用价值/协作健康)。从默认分支提交时间线、合并 PR 演进模式、文档演进、实验/评测文件组织、创新点五个角度回答「这个科研项目怎么一步步长成、值得引用/复现到哪一步」。当用户提到「科研分析」「项目洞悉」「谱系」「lineage」「仓库画像」「可复现性」「引用价值」「科研洞悉」「research insight」时触发。"
|
||||
metadata:
|
||||
requires:
|
||||
bins: ["gitlink-cli"]
|
||||
cliHelp: "gitlink-cli repo --help"
|
||||
python: ["scripts/research/requirements.txt"]
|
||||
cliHelp: "python scripts/research/lineage.py --help"
|
||||
scenario: "S1"
|
||||
---
|
||||
|
||||
# gitlink-research-insight(科研仓库画像)
|
||||
|
|
@ -16,6 +18,80 @@ metadata:
|
|||
|
||||
> **前置条件:** 先读 [`../gitlink-shared/SKILL.md`](../gitlink-shared/SKILL.md)、[`../gitlink-health/SKILL.md`](../gitlink-health/SKILL.md)(工程健康度指标,本 Skill 在其基础上加科研视角)。
|
||||
|
||||
---
|
||||
|
||||
## 子赛题四 · 场景 S1:仓库级科研项目洞悉(lineage 谱系分析)
|
||||
|
||||
> 子赛题四「应用 GitLink 辅助科研」· 场景 **S1 仓库级科研项目洞悉** ——
|
||||
> 本 Skill 的可复现算法实现,技术栈 = **Go 出数据(gitlink-cli + collect.py)+ Python 做算法(lineage.py)**。
|
||||
|
||||
### 何时使用(S1)
|
||||
- 想快速看懂一个陌生科研代码仓库「是怎么一步步长成现在这样的」(提交谱系、合并节奏)。
|
||||
- 评估一个开源科研项目「值得引用/复现到哪一步」「哪些合并是关键创新/里程碑」。
|
||||
- 为论文综述/相关工作梳理某仓库的演进脉络(提交时间线 + 高影响合并 + 文档演进 + 实验组织)。
|
||||
|
||||
### 前置条件
|
||||
1. 已 `gitlink-cli auth login`(Token 7 天有效)。
|
||||
2. 已 `pip install -r scripts/research/requirements.txt`(本场景实际只用标准库,无重型依赖)。
|
||||
3. 目标仓库存在且有默认分支提交历史(`repo +info` 能取到 `default_branch`)。
|
||||
|
||||
### 工作流(S1:lineage.py,Go 出数据 + Python 做算法)
|
||||
1. **取数**(collect.py 薄封装 gitlink-cli):
|
||||
- `c.repo_info`(默认分支)、`c.commits(ref=默认分支, max_pages=10)`(提交历史)
|
||||
- `c.prs(state=merged)`(已合并 PR)、`c.tree` / `c.tree(path='docs')`(仓库树 + 文档树)、`c.readme`。
|
||||
2. **算法**(lineage.py 纯函数,已单测覆盖,不联网):
|
||||
- `is_experiment_file(path)`:命中 experiment*/benchmark*/eval*/tests?/data/ → 科研产物文件。
|
||||
- `is_doc_file(path)`:*.md / docs/* → 文档。
|
||||
- `build_branch_map(commits, default_branch)`:单分支简化 → `[{name, commits, last_active, is_default}]`。
|
||||
- `pr_merge_patterns(merged_prs)`:`[{number, title, status, merged_time, changed_files}]`,按合并时间升序。
|
||||
- `doc_evolution(tree)`:docs/*.md 的 `{file, last_date(近似)}`。
|
||||
- `innovation_points(merged_prs, commits)`:高影响合并(改文件多 / 合入默认分支 / 含里程碑关键词)→ `[{description, evidence, category}]`。
|
||||
3. **产物**:
|
||||
- `lineage.json`(commit_timeline / branch_map / pr_merge_patterns / doc_evolution / experiment_files / innovation_points)
|
||||
- `report.md`(中文洞悉报告)
|
||||
- `branch_graph.mmd`(Mermaid **gitGraph** 分支演进图)
|
||||
|
||||
### 命令(S1)
|
||||
```bash
|
||||
# 默认输出到 stdout(JSON)
|
||||
python scripts/research/lineage.py --owner mindspore-Ecosystem --repo mindspore
|
||||
|
||||
# 输出三件产物到目录
|
||||
python scripts/research/lineage.py --owner <OWNER> --repo <REPO> --branches-limit 5 --out ./out
|
||||
|
||||
# 可复现脚本(封装了上述流程)
|
||||
bash skills/gitlink-research-insight/examples/research-insight-workflow.sh <OWNER> <REPO> [OUT_DIR]
|
||||
```
|
||||
|
||||
### 输出结构(lineage.json,S1)
|
||||
```json
|
||||
{
|
||||
"scenario": "S1_repository_research_insight",
|
||||
"repo": "owner/repo", "default_branch": "master",
|
||||
"commit_timeline": [{"date": "2024-05-01", "count": 12}],
|
||||
"branch_map": [{"name": "master", "commits": 320, "last_active": "2024-06-01", "is_default": true}],
|
||||
"pr_merge_patterns": [{"number": 2, "title": "...", "status": 1, "merged_time": "2024-06-01", "changed_files": 25}],
|
||||
"doc_evolution": [{"file": "guide.md", "last_date": "2024-03-01"}],
|
||||
"experiment_files": ["benchmark/eval.py", "tests/test_model.py"],
|
||||
"innovation_points": [{"description": "...", "evidence": "PR #2 ...", "category": "大规模重构/新特性"}],
|
||||
"meta": {"commit_count": 320, "merged_pr_count": 9, "doc_count": 5, "experiment_file_count": 8}
|
||||
}
|
||||
```
|
||||
|
||||
### 验证(S1)
|
||||
已在真实科研仓库 **`mindspore-Ecosystem/mindspore`**(default_branch=master;issue≈20346;PR=9;贡献者=6)验证:
|
||||
默认分支提交时间线、合并 PR 演进模式、docs 文档清单、benchmark/tests 等实验文件均正确识别;
|
||||
高影响合并被标为「大规模重构/新特性」或「特性引入」创新点。
|
||||
|
||||
### 兼容性(S1)
|
||||
兼容 Claude Code 等 AI Agent:本 SKILL.md 即为 Agent 编排依据,
|
||||
Agent 可直接调上述命令并把 `lineage.json` / `report.md` / `branch_graph.mmd` 读回做进一步解读与文案化。
|
||||
单测 `python scripts/research/test_lineage.py` 全部离线通过(不联网、不调 gitlink-cli)。
|
||||
|
||||
---
|
||||
|
||||
|
||||
|
||||
## 定位:科研辅助,与 gitlink-health 的分工
|
||||
|
||||
| Skill | 视角 | 核心问题 |
|
||||
|
|
|
|||
|
|
@ -0,0 +1,46 @@
|
|||
#!/usr/bin/env bash
|
||||
# research-insight-workflow.sh — S1 仓库级科研项目洞悉 · 可复现执行脚本
|
||||
# 子赛题四「应用 GitLink 辅助科研」交付物之一(Go 出数据 + Python 做算法)
|
||||
#
|
||||
# 用法: bash research-insight-workflow.sh <OWNER> <REPO> [OUT_DIR] [BRANCHES_LIMIT]
|
||||
# 示例: bash research-insight-workflow.sh mindspore-Ecosystem mindspore ./out 5
|
||||
set -euo pipefail
|
||||
|
||||
OWNER="${1:?用法: $0 <OWNER> <REPO> [OUT_DIR] [BRANCHES_LIMIT]}"
|
||||
REPO="${2:?缺少 REPO}"
|
||||
OUT_DIR="${3:-./research-insight-output}"
|
||||
BRANCHES_LIMIT="${4:-5}"
|
||||
|
||||
# 定位仓库根(脚本位于 skills/gitlink-research-insight/examples/)
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
REPO_ROOT="$(cd "$SCRIPT_DIR/../../.." && pwd)"
|
||||
LINEAGE="$REPO_ROOT/scripts/research/lineage.py"
|
||||
|
||||
# 本地 Windows 开发可设 GITLINK_CLI=./gitlink-cli.exe;Linux/容器走 PATH 默认值
|
||||
: "${GITLINK_CLI:=gitlink-cli}"
|
||||
|
||||
echo "==> 目标仓库: $OWNER/$REPO"
|
||||
echo "==> CLI: $GITLINK_CLI"
|
||||
echo "==> 分支地图上限=$BRANCHES_LIMIT"
|
||||
|
||||
mkdir -p "$OUT_DIR"
|
||||
GITLINK_CLI="$GITLINK_CLI" python "$LINEAGE" \
|
||||
--owner "$OWNER" --repo "$REPO" \
|
||||
--branches-limit "$BRANCHES_LIMIT" \
|
||||
--out "$OUT_DIR"
|
||||
|
||||
echo
|
||||
echo "==> 产物:"
|
||||
ls -1 "$OUT_DIR"
|
||||
echo
|
||||
echo "==> 洞悉摘要预览:"
|
||||
python -c "
|
||||
import json
|
||||
d=json.load(open('$OUT_DIR/lineage.json',encoding='utf-8'))
|
||||
m=d['meta']
|
||||
print(f\"默认分支: {d['default_branch']} | 提交 {m['commit_count']} 条 | 合并 PR {m['merged_pr_count']} 个\")
|
||||
print(f\"文档 {m['doc_count']} 个 | 实验/评测文件 {m['experiment_file_count']} 个\")
|
||||
print('创新/里程碑点:')
|
||||
for i,it in enumerate(d['innovation_points'][:8],1):
|
||||
print(f\" {i}. [{it['category']}] {it['description']}\")
|
||||
"
|
||||
|
|
@ -0,0 +1,72 @@
|
|||
---
|
||||
name: gitlink-research-progress
|
||||
version: 1.0.0
|
||||
description: "科研进度智能跟踪与预警(子赛题四·S5):统计科研仓库本周/上周的提交、Issue、PR 活跃度,结合里程碑进度,用阈值规则产出风险预警(stale issue / stale PR / 逾期里程碑 / 低活跃 / bus factor),生成科研进度周报。当用户要做项目周报、进度跟踪、风险预警、里程碑监控时触发。"
|
||||
metadata:
|
||||
requires:
|
||||
bins: ["gitlink-cli"]
|
||||
cliHelp: "python scripts/research/report.py --help"
|
||||
scenario: "S5"
|
||||
---
|
||||
|
||||
# gitlink-research-progress — 科研进度智能跟踪与预警
|
||||
|
||||
> 子赛题四「应用 GitLink 辅助科研」· 场景 **S5 科研进度智能跟踪与预警**
|
||||
|
||||
## 何时使用
|
||||
|
||||
- 课题组负责人想每周自动生成「科研项目进度周报」。
|
||||
- 想及时发现项目停滞(低活跃)、Issue/PR 堆积、里程碑逾期、单点依赖(bus factor)。
|
||||
- 为科研项目做里程碑监控与风险预警。
|
||||
|
||||
## 前置条件
|
||||
|
||||
1. 已 `gitlink-cli auth login`。
|
||||
2. 本场景为纯标准库实现,无需额外 pip 依赖(`scripts/research/report.py`)。
|
||||
|
||||
## 工作流
|
||||
|
||||
算法由 `scripts/research/report.py` 实现(Go 出数据 + Python 做统计/预警):
|
||||
|
||||
1. **取数**:`commits`(Raw API)/ `issue +list --state all` / `pr +list --state all` / `milestone +list` / `repo +contributors`。
|
||||
2. **周统计**:按时间窗口把提交/Issue/PR 划入「本周 [now-7d, now]」与「上周 [now-14d, now-7d)」,统计新增/关闭/stale/活跃贡献者。
|
||||
3. **里程碑进度**:按 `milestone_name` 归集 Issue 的 open/closed,算完成率与逾期。
|
||||
4. **风险预警**(阈值规则):
|
||||
- `low_activity`:本周提交 < 3
|
||||
- `bus_factor`:单一贡献者占本周提交 > 50% 且活跃贡献者 ≤ 2(critical)
|
||||
- `stale_issue`:开放 Issue 超 30 天无活动(≥5 触发,≥20 升级 critical)
|
||||
- `stale_pr`:开放 PR 超 14 天未 review(≥1 触发)
|
||||
- `overdue_milestone`:未关闭里程碑已过 due_date(critical)
|
||||
5. **趋势**:本周 vs 上周 commit 环比,给出 increasing/stable/decreasing。
|
||||
6. **产物**:`report.json`(结构化)+ `weekly_report.md`(中文周报:活动对比表 + 里程碑表 + 风险预警表)。
|
||||
|
||||
## 命令
|
||||
|
||||
```bash
|
||||
python scripts/research/report.py --owner mindspore-Ecosystem --repo mindspore --out ./out
|
||||
python scripts/research/report.py --owner O --repo R # 仅打印 JSON
|
||||
|
||||
# 可复现脚本
|
||||
bash skills/gitlink-research-progress/examples/progress-report-workflow.sh <OWNER> <REPO> [OUT_DIR]
|
||||
```
|
||||
|
||||
## 输出结构(report.json)
|
||||
|
||||
```json
|
||||
{
|
||||
"scenario": "S5_progress_tracking", "repo": "owner/repo",
|
||||
"week_stats": {"this_week": {...}, "last_week": {...}, "window": {...}},
|
||||
"trend": {"commit_delta_pct": 12.5, "activity_level": "increasing"},
|
||||
"milestones": [{"name":"v1.0","open":2,"closed":1,"completion_pct":33.3,"overdue":true}],
|
||||
"risk_warnings": [{"level":"critical","type":"bus_factor","message":"...","suggestion":"..."}]
|
||||
}
|
||||
```
|
||||
|
||||
## 验证
|
||||
|
||||
已在真实科研仓库 **`mindspore-Ecosystem/mindspore`** 上验证取数与统计口径;
|
||||
11 个纯单元测试覆盖时间解析、周分桶、stale 判定、bus factor、里程碑逾期、趋势计算。
|
||||
|
||||
## 兼容性
|
||||
|
||||
兼容 Claude Code 等 AI Agent:本 SKILL.md 为编排依据,Agent 调上述命令并把周报读回做解读与跟进建议。
|
||||
|
|
@ -0,0 +1,35 @@
|
|||
#!/usr/bin/env bash
|
||||
# progress-report-workflow.sh — S5 科研进度周报 · 可复现执行脚本
|
||||
# 子赛题四「应用 GitLink 辅助科研」交付物之一
|
||||
#
|
||||
# 用法: bash progress-report-workflow.sh <OWNER> <REPO> [OUT_DIR]
|
||||
# 示例: bash progress-report-workflow.sh mindspore-Ecosystem mindspore ./out
|
||||
set -euo pipefail
|
||||
|
||||
OWNER="${1:?用法: $0 <OWNER> <REPO> [OUT_DIR]}"
|
||||
REPO="${2:?缺少 REPO}"
|
||||
OUT_DIR="${3:-./progress-output}"
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
REPO_ROOT="$(cd "$SCRIPT_DIR/../../.." && pwd)"
|
||||
REPORT="$REPO_ROOT/scripts/research/report.py"
|
||||
|
||||
: "${GITLINK_CLI:=gitlink-cli}"
|
||||
|
||||
echo "==> 目标仓库: $OWNER/$REPO (CLI: $GITLINK_CLI)"
|
||||
mkdir -p "$OUT_DIR"
|
||||
GITLINK_CLI="$GITLINK_CLI" python "$REPORT" --owner "$OWNER" --repo "$REPO" --out "$OUT_DIR"
|
||||
|
||||
echo
|
||||
echo "==> 产物:"; ls -1 "$OUT_DIR"
|
||||
echo
|
||||
echo "==> 摘要:"
|
||||
python -c "
|
||||
import json
|
||||
d=json.load(open('$OUT_DIR/report.json',encoding='utf-8'))
|
||||
tw=d['week_stats']['this_week']; tr=d['trend']
|
||||
print(f\"本周提交 {tw['commits']},活跃贡献者 {tw['contributors_active']},趋势 {tr['activity_level']}({tr['commit_delta_pct']}%)\")
|
||||
print(f\"风险预警 {len(d['risk_warnings'])} 条:\")
|
||||
for w in d['risk_warnings']:
|
||||
print(f\" [{w['level']}] {w['type']}: {w['message']}\")
|
||||
"
|
||||
|
|
@ -0,0 +1,84 @@
|
|||
---
|
||||
name: gitlink-research-visual
|
||||
version: 1.0.0
|
||||
description: "科研成果可视化沉淀(子赛题四·S6):把一个科研仓库的开发时间线(commit/issue/pr 周粒度趋势)、贡献者×周热力图、语言占比饼图、里程碑甘特沉淀成一张可交互的 HTML 报告,并从 README/提交信息中抽取论文引用(arXiv/DOI)、按目录分类仓库产物(论文/数据集/模型/基准)。当用户提到「科研成果可视化」「可视化沉淀」「开发节奏」「贡献热力图」「论文引用抽取」「产物分类」「research visualization」时触发。"
|
||||
metadata:
|
||||
requires:
|
||||
bins: ["gitlink-cli"]
|
||||
python: ["scripts/research/requirements.txt"]
|
||||
cliHelp: "gitlink-cli api --help"
|
||||
scenario: "S6"
|
||||
---
|
||||
|
||||
# gitlink-research-visual — 科研成果可视化沉淀
|
||||
|
||||
> 子赛题四「应用 GitLink 辅助科研」· 场景 **S6 科研成果可视化沉淀**
|
||||
|
||||
## 何时使用
|
||||
|
||||
- 想把一个科研代码仓库的「成果产出」沉淀成可视化报告(给导师/合作者/项目主页展示)。
|
||||
- 想看开发节奏:提交/Issue/PR 的周粒度趋势、贡献者活跃热力图。
|
||||
- 想自动整理一份仓库里「能被引用的源头」清单:论文 arXiv/DOI 链接、数据集/模型/基准产物。
|
||||
- 给开源科研项目做年度总结、阶段性汇报、成果网页的数据底座。
|
||||
|
||||
## 前置条件
|
||||
|
||||
1. 已 `gitlink-cli auth login`(Token 7 天有效)。
|
||||
2. 已 `pip install -r scripts/research/requirements.txt`;本场景核心算法仅用标准库,
|
||||
生成交互 HTML 需 `plotly>=5.18`(已列入 requirements)。**未安装 plotly 时优雅降级**:仅写 `visual.json` + `report.md` 并提示。
|
||||
3. 目标仓库存在且有提交历史(时间线/热力图的数据来源)。
|
||||
|
||||
## 工作流
|
||||
|
||||
本 Skill 的算法由 `scripts/research/visual.py` 实现(Go 出数据 + Python 做可视化):
|
||||
|
||||
1. **取数**(`collect()`):`commits`(Raw API 分页,取够 ~weeks 周,max_pages=10)+ `issue +list` + `pr +list` + `milestone +list` + `repo +languages` + `repo +contributors` + `repo +readme` + `repo +tree`。
|
||||
2. **周分桶**(`bin_weekly`):把 commit/issue/pr 按时间戳对齐到 ISO 周,分入「最近 weeks 周」的桶里,得到时间线趋势数据。
|
||||
3. **贡献热力图**(`contribution_heatmap`):top 贡献者 × 周桶的提交数矩阵,定位核心贡献者与活跃周期。
|
||||
4. **论文引用抽取**(`extract_paper_links`):正则匹配 arXiv(`arxiv.org/abs|pdf/`、`arXiv:`)与 DOI(`doi.org/`、裸 `10.xxxx/`),去重并带上下文片段。
|
||||
5. **产物分类**(`classify_artifacts`):按仓库目录路径把文件归入 paper / dataset / model / benchmark 四类。
|
||||
6. **渲染**:单一交互 HTML(plotly 多子图:时间线折线 + 贡献热力 + 语言饼图 + 里程碑甘特),同时输出原始数据 bundle `visual.json`(供网页前端二次开发)+ `report.md`(中文摘要)。
|
||||
|
||||
## 命令
|
||||
|
||||
```bash
|
||||
# 默认输出到 stdout(JSON)
|
||||
python scripts/research/visual.py --owner mindspore-Ecosystem --repo mindspore
|
||||
|
||||
# 输出可视化三件套到目录(visual.html + visual.json + report.md)
|
||||
python scripts/research/visual.py --owner <OWNER> --repo <REPO> --weeks 26 --out ./out
|
||||
|
||||
# 可复现脚本(封装了上述流程)
|
||||
bash skills/gitlink-research-visual/examples/research-visual-workflow.sh <OWNER> <REPO> [OUT_DIR] [WEEKS]
|
||||
```
|
||||
|
||||
## 输出结构(visual.json)
|
||||
|
||||
```json
|
||||
{
|
||||
"scenario": "S6_research_visualization",
|
||||
"repo": "owner/repo",
|
||||
"weeks": 26,
|
||||
"timeline": {"labels": ["2026-W01", "..."], "commits": [...], "issues": [...], "prs": [...]},
|
||||
"heatmap": {"users": ["alice", "..."], "weeks": [...], "matrix": [[0,1,...], ...]},
|
||||
"languages": {"Python": "99.7%"},
|
||||
"milestones": [{"title": "v1.0", "start": 1718000000, "due": 1720000000}],
|
||||
"paper_links": [{"source_text_snippet": "see arxiv...", "target": "https://arxiv.org/abs/2401.00012", "type": "arxiv"}],
|
||||
"artifacts": [{"path": "data/train.csv", "category": "dataset", "name": "train.csv"}],
|
||||
"artifact_summary": {"paper": 3, "dataset": 5, "model": 2, "benchmark": 1},
|
||||
"meta": {"commit_count": 800, "issue_count": 20346, "pr_count": 9, "milestone_count": 4, "contributor_count": 6}
|
||||
}
|
||||
```
|
||||
|
||||
## 验证
|
||||
|
||||
已在真实科研仓库 **`mindspore-Ecosystem/mindspore`**(default_branch=master;issue≈20346;PR=9;贡献者=6)上验证:
|
||||
周分桶正确对齐 ISO 周(无 plotly 时优雅降级为 `visual.json` + `report.md`);
|
||||
时间线/热力图数据由 `bin_weekly`/`contribution_heatmap` 纯函数产出,离线单测 `python scripts/research/test_visual.py` 全部通过(33 个用例,覆盖周分桶/热力矩阵/论文链接抽取/产物分类)。
|
||||
|
||||
## 兼容性
|
||||
|
||||
兼容 Claude Code 等 AI Agent:本 SKILL.md 即为 Agent 编排依据,
|
||||
Agent 可直接调上述命令并把产物(HTML/JSON/Markdown)读回做进一步解读与文案化。
|
||||
纯函数算法(`bin_weekly`/`contribution_heatmap`/`extract_paper_links`/`classify_artifacts`)与取数层解耦,
|
||||
可在不联网、不调 gitlink-cli 的前提下被单测与复用。
|
||||
|
|
@ -0,0 +1,48 @@
|
|||
#!/usr/bin/env bash
|
||||
# research-visual-workflow.sh — S6 科研成果可视化沉淀 · 可复现执行脚本
|
||||
# 子赛题四「应用 GitLink 辅助科研」交付物之一
|
||||
#
|
||||
# 用法: bash research-visual-workflow.sh <OWNER> <REPO> [OUT_DIR] [WEEKS]
|
||||
# 示例: bash research-visual-workflow.sh mindspore-Ecosystem mindspore ./out 26
|
||||
set -euo pipefail
|
||||
|
||||
OWNER="${1:?用法: $0 <OWNER> <REPO> [OUT_DIR] [WEEKS]}"
|
||||
REPO="${2:?缺少 REPO}"
|
||||
OUT_DIR="${3:-./research-visual-output}"
|
||||
WEEKS="${4:-26}"
|
||||
|
||||
# 定位仓库根(脚本位于 skills/gitlink-research-visual/examples/)
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
REPO_ROOT="$(cd "$SCRIPT_DIR/../../.." && pwd)"
|
||||
VISUAL="$REPO_ROOT/scripts/research/visual.py"
|
||||
|
||||
# 本地 Windows 开发可设 GITLINK_CLI=./gitlink-cli.exe;Linux/容器走 PATH 默认值
|
||||
: "${GITLINK_CLI:=gitlink-cli}"
|
||||
|
||||
echo "==> 目标仓库: $OWNER/$REPO"
|
||||
echo "==> CLI: $GITLINK_CLI"
|
||||
echo "==> 回溯周数: $WEEKS"
|
||||
|
||||
mkdir -p "$OUT_DIR"
|
||||
GITLINK_CLI="$GITLINK_CLI" python "$VISUAL" \
|
||||
--owner "$OWNER" --repo "$REPO" \
|
||||
--weeks "$WEEKS" \
|
||||
--out "$OUT_DIR"
|
||||
|
||||
echo
|
||||
echo "==> 产物:"
|
||||
ls -1 "$OUT_DIR"
|
||||
echo
|
||||
echo "==> 成果概览:"
|
||||
python -c "
|
||||
import json
|
||||
d=json.load(open('$OUT_DIR/visual.json',encoding='utf-8'))
|
||||
m=d['meta']; a=d['artifact_summary']
|
||||
print(f\"仓库: {d['repo']} 回溯 {d['weeks']} 周\")
|
||||
print(f\"活跃度: commits={m['commit_count']} issues={m['issue_count']} prs={m['pr_count']} 贡献者={m['contributor_count']}\")
|
||||
print(f\"产物分类: paper={a['paper']} dataset={a['dataset']} model={a['model']} benchmark={a['benchmark']}\")
|
||||
print(f\"抽取论文引用: {len(d['paper_links'])} 条\")
|
||||
tl=d['timeline']
|
||||
peak=max(tl['commits']) if tl['commits'] else 0
|
||||
print(f\"开发节奏(最近窗口): 峰值 {peak} 提交/周\")
|
||||
"
|
||||
Loading…
Reference in New Issue