diff --git a/CHANGELOG.md b/CHANGELOG.md index 1ef51df..4a68f36 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -23,6 +23,9 @@ This project uses [Semantic Versioning](https://semver.org/spec/v2.0.0.html). - Multi-channel packaging via GoReleaser: Homebrew Cask (`openhat-security/homebrew-tap`), Scoop (`openhat-security/scoop-bucket`), `.deb`/`.rpm` (nfpm), apt + dnf repos on GitHub Pages (`openhat-security/packages`), AUR `runhug-bin` PKGBUILD, winget manifest generator, npm under `packaging/npm`. - Deploy-time **sampling best-practice** step (`--sampling recommended|none`, `--set key=value`): family table + quant-aware nudge (GGUF/AWQ/GPTQ) + Hub `generation_config.json`. Applied server-side on GCP llama-server; request-time via `runhug run` / Claude bridge on RunPod. - README / hero: deploy **any** model, including heretic/abliterated builds (`runhug heretic make`). +- `runhug gpu list|set|clear|show` — vendored NVIDIA GPU index (Jr23xd23/gpu-database), `--filter all|local|runpod|gcp`, `--sort best|cheapest|value|vram|name`; preference wired into deploy / recommend / GCP. `gpus` aliases `gpu list`. +- `runhug local run` — start local runtime if needed, then chat. Host GPU probe via `nvidia-smi` / Apple Silicon for `--filter local`. +- `runhug gpu update` — refresh shipped NVIDIA + AMD + GCP catalogs into `~/.config/runhug/gpudb/` (HF index, curated GCP list, optional `gcloud` merge / GitHub Release assets `gpudb-*.json`). ### Changed - `assets/hero.svg`: Claude Code, OpenCode, RunPod, and GCP marks; any-model / heretic copy. diff --git a/README.md b/README.md index cf3ffa3..d4ce0bc 100644 --- a/README.md +++ b/README.md @@ -39,7 +39,7 @@ runhug start opencode # Use your model with OpenCode. **QUEUE** OpenAI base: `https://api.runpod.ai/v2/{id}/openai/v1` **Load balancer** OpenAI base: `https://{id}.api.runpod.ai/v1` -Other useful commands: `inspect`, `update` / `update --packs`, `gpus`, `url`, `status`, `local add` / `local setup`, `config get|set`. Env: `RUNPOD_API_KEY`, `HF_TOKEN`, `RUNHUG_CONFIG`, `NO_COLOR`. +Other useful commands: `inspect`, `update` / `update --packs`, `gpu list|set|clear|update`, `url`, `status`, `local add|start|stop|run|setup`, `config get|set`. Env: `RUNPOD_API_KEY`, `HF_TOKEN`, `RUNHUG_CONFIG`, `NO_COLOR`. ## Install diff --git a/internal/cli/deploy.go b/internal/cli/deploy.go index eef76e9..cf009bb 100644 --- a/internal/cli/deploy.go +++ b/internal/cli/deploy.go @@ -94,7 +94,11 @@ func cmdDeploy(args []string) error { if err != nil { return err } - choice, err := runpod.Pick(gpus, est.RequiredGB, *gpu, *gpuCount) + preferGPU := applySavedRunpodGPU(*gpu) + if preferGPU != "" && strings.TrimSpace(*gpu) == "" { + fmt.Fprintf(os.Stderr, "%s using saved GPU preference %s\n", dim("note:"), preferGPU) + } + choice, err := runpod.Pick(gpus, est.RequiredGB, preferGPU, *gpuCount) if err != nil { return err } diff --git a/internal/cli/gcp.go b/internal/cli/gcp.go index eec43ac..0a9bae2 100644 --- a/internal/cli/gcp.go +++ b/internal/cli/gcp.go @@ -110,7 +110,7 @@ func cmdGCPDeploy(args []string) error { project := fs.String("project", "", "GCP project id (no hardcoded default)") zone := fs.String("zone", "", "GCE zone") region := fs.String("region", "", "GCE region") - gpu := fs.String("gpu", "", "L4 (default) or T4") + gpu := fs.String("gpu", "", "L4 (default) or T4; default: gpu preference if gcp") idle := fs.Int("idle-timeout", 600, "seconds before stop-on-idle") keepUpFlag := fs.Bool("keep-up", false, "disable stop-on-idle (bills until gcp stop)") name := fs.String("name", "", "instance name") @@ -212,9 +212,13 @@ func cmdGCPDeploy(args []string) error { } keepUp := *keepUpFlag + preferGPU := applySavedGCPGPU(*gpu) + if preferGPU != "" && strings.TrimSpace(*gpu) == "" { + fmt.Fprintf(os.Stderr, "%s using saved GPU preference %s\n", dim("note:"), preferGPU) + } if !*dry && !*yes && promptOK() { rate := gcp.SpotHourlyL4 - if strings.EqualFold(strings.TrimSpace(*gpu), "T4") { + if strings.EqualFold(strings.TrimSpace(preferGPU), "T4") { rate = gcp.SpotHourlyT4 } warn := fmt.Sprintf("Keep Spot VM up with NO stop-on-idle? Bills ~$%.2f/hr until you run runhug gcp stop", rate) @@ -249,7 +253,7 @@ func cmdGCPDeploy(args []string) error { Name: *name, ModelID: modelID, GGUFFile: ggufFile, - GPU: *gpu, + GPU: preferGPU, IdleSeconds: *idle, KeepUp: keepUp, DiskGB: *disk, diff --git a/internal/cli/gpu.go b/internal/cli/gpu.go new file mode 100644 index 0000000..aca2dbc --- /dev/null +++ b/internal/cli/gpu.go @@ -0,0 +1,413 @@ +package cli + +import ( + "bufio" + "context" + "fmt" + "os" + "strconv" + "strings" + "time" + + "github.com/adamsiwiec1/runhug/internal/config" + "github.com/adamsiwiec1/runhug/internal/gpudb" +) + +func cmdGPU(args []string) error { + if len(args) == 0 { + return cmdGPUList(nil) + } + switch strings.ToLower(args[0]) { + case "list", "ls": + return cmdGPUList(args[1:]) + case "set": + return cmdGPUSet(args[1:]) + case "clear", "unset": + return cmdGPUClear(args[1:]) + case "show", "get", "status": + return cmdGPUShow(args[1:]) + case "update", "refresh": + return cmdGPUUpdate(args[1:]) + case "help", "-h", "--help": + printGPUUsage() + return nil + default: + // Treat unknown first token as list query for convenience: `gpu 4090` + if strings.HasPrefix(args[0], "-") { + return cmdGPUList(args) + } + return fmt.Errorf("unknown gpu command %q (list|set|clear|show)", args[0]) + } +} + +func printGPUUsage() { + fmt.Fprintln(os.Stdout, "usage:") + fmt.Fprintln(os.Stdout, " runhug gpu list [--filter all|local|amd|nvidia|runpod|gcp] [--sort best|cheapest|value|vram|name] [--query Q]") + fmt.Fprintln(os.Stdout, " runhug gpu set [NAME|POOL|L4|T4] [--filter …]") + fmt.Fprintln(os.Stdout, " runhug gpu clear") + fmt.Fprintln(os.Stdout, " runhug gpu show") + fmt.Fprintln(os.Stdout, " runhug gpu update [--nvidia|--amd|--gcp] refresh catalogs (like `runhug update` for models)") + fmt.Fprintln(os.Stdout, " runhug gpus … alias for gpu list") +} + +func cmdGPUUpdate(args []string) error { + fs := newFlagSet("gpu update") + nvidia := fs.Bool("nvidia", false, "refresh NVIDIA hardware index only") + amd := fs.Bool("amd", false, "refresh AMD hardware index only") + gcpOnly := fs.Bool("gcp", false, "refresh GCP accelerator catalog only") + preferRelease := fs.Bool("release", true, "prefer GitHub Release assets when published (gpudb-*.json)") + if err := parseFlags(fs, args); err != nil { + return err + } + any := *nvidia || *amd || *gcpOnly + opts := gpudb.UpdateOptions{ + NVIDIA: *nvidia || !any, + AMD: *amd || !any, + GCP: *gcpOnly || !any, + PreferRelease: *preferRelease, + } + if any { + opts.NVIDIA, opts.AMD, opts.GCP = *nvidia, *amd, *gcpOnly + } + + heading(os.Stdout, "Update GPU catalogs") + fmt.Fprintln(os.Stdout, dim("Shipped with the CLI; updates land in ~/.config/runhug/gpudb/ (same idea as model index packs).")) + fmt.Fprintln(os.Stdout) + + ctx, cancel := context.WithTimeout(context.Background(), 3*time.Minute) + defer cancel() + res, err := gpudb.Update(ctx, opts) + if err != nil { + return err + } + if opts.NVIDIA { + printKV(os.Stdout, "nvidia", fmt.Sprintf("%d SKUs from %s", res.NVIDIACount, res.NVIDIAFrom)) + } + if opts.AMD { + printKV(os.Stdout, "amd", fmt.Sprintf("%d SKUs from %s", res.AMDCount, res.AMDFrom)) + } + if opts.GCP { + printKV(os.Stdout, "gcp", fmt.Sprintf("%d accelerators from %s", res.GCPCount, res.GCPFrom)) + } + printKV(os.Stdout, "cache", res.CacheDir) + fmt.Fprintln(os.Stdout) + fmt.Fprintln(os.Stdout, dim("List: runhug gpu list --query MI300 or --filter local")) + return nil +} + +func cmdGPUList(args []string) error { + fs := newFlagSet("gpu list") + filter := fs.String("filter", "all", "all | local | amd | nvidia | runpod | gcp") + sortKey := fs.String("sort", "best", "best | cheapest | value | vram | name") + query := fs.String("query", "", "substring filter on name/pool") + fs.StringVar(query, "q", "", "alias for --query") + limit := fs.Int("limit", 0, "max rows (0 = all)") + minVRAM := fs.Float64("min-vram", 0, "minimum VRAM GB") + asJSON := fs.Bool("json", false, "print JSON") + if err := parseFlags(fs, args); err != nil { + return err + } + if fs.NArg() > 0 && strings.TrimSpace(*query) == "" { + *query = strings.Join(fs.Args(), " ") + } + + isLocal := strings.EqualFold(*filter, "local") + var onMachine []gpuRow + if isLocal { + specs, err := gpudb.Load() + if err != nil { + return err + } + onMachine = detectedLocalRows(specs) + } + + rows, err := buildGPURows(*filter, *query) + if err != nil { + return err + } + if *minVRAM > 0 { + filtered := rows[:0] + for _, r := range rows { + if r.MemoryGB >= *minVRAM { + filtered = append(filtered, r) + } + } + rows = filtered + } + sortGPURows(rows, *sortKey) + catalogAll := rows + yRank, yTotal, yRow := yoursRank(rows) + if *limit > 0 && len(rows) > *limit { + // Keep the user's GPU visible even when limiting. + if yRank > *limit && yRank > 0 { + kept := append([]gpuRow{}, rows[:*limit-1]...) + kept = append(kept, rows[yRank-1]) + rows = kept + } else { + rows = rows[:*limit] + } + } + + if *asJSON { + if isLocal { + return writeJSON(localListPayload{ + OnThisMachine: onMachine, + Catalog: catalogAll, + YoursRank: yRank, + YoursTotal: yTotal, + }) + } + type allPayload struct { + GPUs []gpuRow `json:"gpus"` + YoursRank int `json:"yours_rank,omitempty"` + YoursTotal int `json:"yours_total,omitempty"` + Yours *gpuRow `json:"yours,omitempty"` + } + p := allPayload{GPUs: catalogAll, YoursRank: yRank, YoursTotal: yTotal} + if yRank > 0 { + r := yRow + p.Yours = &r + } + return writeJSON(p) + } + + heading(os.Stdout, "GPUs") + printKV(os.Stdout, "filter", *filter) + printKV(os.Stdout, "sort", *sortKey) + if pref := savedGPUPreference(); pref != nil { + printKV(os.Stdout, "preference", fmt.Sprintf("%s %s (%s)", pref.Provider, pref.Key, pref.Name)) + } + if yRank > 0 { + extra := "" + if yRow.FP16 > 0 { + extra = fmt.Sprintf(" · FP16≈%.0f", yRow.FP16) + } + printKV(os.Stdout, "yours", fmt.Sprintf("%s — rank #%d of %d by %s%s", + yRow.Name, yRank, yTotal, *sortKey, extra)) + } + fmt.Fprintln(os.Stdout) + + if isLocal { + fmt.Fprintln(os.Stdout, bold("On this machine")) + if len(onMachine) == 0 { + fmt.Fprintln(os.Stdout, dim(" (none detected — install NVIDIA drivers / nvidia-smi, or Apple Silicon for unified memory)")) + } else { + printGPUTable(onMachine) + } + fmt.Fprintln(os.Stdout) + fmt.Fprintf(os.Stdout, "%s %s\n", bold("Local catalog"), dim(fmt.Sprintf("(%d SKUs you can run locally)", len(catalogAll)))) + } + + printGPUTableHeader() + if len(rows) == 0 { + fmt.Fprintln(os.Stdout, dim(" (no GPUs matched)")) + if strings.EqualFold(*filter, "runpod") && config.Load().RunpodAPIKey == "" { + fmt.Fprintln(os.Stdout, dim(" tip: run `runhug connect` for live stock; showing offline catalog")) + } + return nil + } + printGPUTableRows(rows) + fmt.Fprintln(os.Stdout) + fmt.Fprintln(os.Stdout, dim("FP16 = TechPowerUp vector TFLOPS (not tensor-peak). TC = tensor core count.")) + if strings.EqualFold(*filter, "all") { + fmt.Fprintln(os.Stdout, dim("PROV - = in the hardware index only (not currently joined to RunPod or GCP).")) + } + if yRank > 0 { + fmt.Fprintln(os.Stdout, dim("STOCK yours = GPU on this machine (Apple FP16 is an estimate for ranking).")) + } + fmt.Fprintln(os.Stdout, dim("Set preference: runhug gpu set Clear: runhug gpu clear")) + return nil +} + +func printGPUTableHeader() { + fmt.Fprintf(os.Stdout, " %s %s %s %s %s %s %s %s %s\n", + dim(padRight("#", 4)), + dim(padRight("NAME", 28)), + dim(padRight("VRAM", 6)), + dim(padRight("FP16", 7)), + dim(padRight("TC", 5)), + dim(padRight("PROV", 7)), + dim(padRight("KEY", 12)), + dim(padRight("$/HR", 6)), + dim("STOCK"), + ) +} + +func printGPUTable(rows []gpuRow) { + printGPUTableHeader() + printGPUTableRows(rows) +} + +func printGPUTableRows(rows []gpuRow) { + for i, r := range rows { + stock := r.Stock + stockOut := padRight(stock, 8) + switch { + case r.Yours || stock == "yours": + stockOut = green(padRight("yours", 8)) + case r.Provider == "runpod" || r.Provider == "gcp": + if r.InStock { + stockOut = green(padRight(stock, 8)) + } else if stock != "" { + stockOut = yellow(padRight(stock, 8)) + } + } + price := "-" + if r.PricePerHr > 0 { + price = fmt.Sprintf("%.2f", r.PricePerHr) + } + fp16 := "-" + if r.FP16 > 0 { + fp16 = fmt.Sprintf("%.0f", r.FP16) + } + tc := "-" + if r.TensorCores > 0 { + tc = strconv.Itoa(r.TensorCores) + } + prov := r.Provider + if prov == "" { + prov = "-" + } + key := r.Key + if key == "" { + key = "-" + } + name := r.Name + if r.Yours { + name = name + " *" + } + fmt.Fprintf(os.Stdout, " %s %s %s %s %s %s %s %s %s\n", + padRight(strconv.Itoa(i+1), 4), + bold(padRight(truncateRunes(name, 28), 28)), + padRight(fmt.Sprintf("%.0f", r.MemoryGB), 6), + padRight(fp16, 7), + padRight(tc, 5), + padRight(prov, 7), + padRight(truncateRunes(key, 12), 12), + padRight(price, 6), + stockOut, + ) + } +} + +func cmdGPUSet(args []string) error { + fs := newFlagSet("gpu set") + filter := fs.String("filter", "all", "all | local | amd | nvidia | runpod | gcp") + sortKey := fs.String("sort", "best", "best | cheapest | value | vram | name") + if err := parseFlags(fs, args); err != nil { + return err + } + want := strings.TrimSpace(strings.Join(fs.Args(), " ")) + rows, err := buildGPURows(*filter, "") + if err != nil { + return err + } + sortGPURows(rows, *sortKey) + if len(rows) == 0 { + return fmt.Errorf("no GPUs for --filter %s", *filter) + } + + var chosen gpuRow + if want == "" { + if !stdinIsTTY() { + return fmt.Errorf("usage: runhug gpu set [--filter …]") + } + // Reuse list display then prompt. + _ = cmdGPUList([]string{"--filter", *filter, "--sort", *sortKey, "--limit", "40"}) + fmt.Fprint(os.Stderr, "Pick # or name: ") + sc := bufio.NewScanner(os.Stdin) + if !sc.Scan() { + return fmt.Errorf("cancelled") + } + want = strings.TrimSpace(sc.Text()) + if want == "" { + return fmt.Errorf("cancelled") + } + if n, err := strconv.Atoi(want); err == nil && n >= 1 && n <= len(rows) { + chosen = rows[n-1] + } else { + chosen, err = resolveGPUPreference(*filter, want) + if err != nil { + return err + } + } + } else if n, err := strconv.Atoi(want); err == nil && n >= 1 && n <= len(rows) { + chosen = rows[n-1] + } else { + chosen, err = resolveGPUPreference(*filter, want) + if err != nil { + return err + } + } + + provider := chosen.Provider + if provider == "" { + // Index-only pick: infer provider from filter when possible. + switch strings.ToLower(*filter) { + case "local", "runpod", "gcp": + provider = strings.ToLower(*filter) + default: + provider = "runpod" // deploy default; user can re-set with --filter + if chosen.Key == "" { + chosen.Key = chosen.Name + } + } + } + if chosen.Key == "" { + chosen.Key = chosen.Name + } + pref := &config.GPUPreference{ + Provider: provider, + Key: chosen.Key, + Name: chosen.Name, + MemoryGB: chosen.MemoryGB, + } + s := config.LoadSettings() + s.GPU = pref + if err := config.SaveSettings(s); err != nil { + return err + } + fmt.Fprintf(os.Stdout, "%s GPU preference %s / %s", green("✓"), bold(pref.Provider), bold(pref.Key)) + if pref.Name != "" && pref.Name != pref.Key { + fmt.Fprintf(os.Stdout, " (%s)", pref.Name) + } + fmt.Fprintln(os.Stdout) + if pref.MemoryGB > 0 { + printKV(os.Stdout, "vram", fmt.Sprintf("%.0f GB", pref.MemoryGB)) + } + return nil +} + +func cmdGPUClear(args []string) error { + _ = args + s := config.LoadSettings() + if s.GPU == nil { + fmt.Fprintln(os.Stdout, dim("No GPU preference set.")) + return nil + } + s.GPU = nil + if err := config.SaveSettings(s); err != nil { + return err + } + fmt.Fprintf(os.Stdout, "%s GPU preference cleared\n", green("✓")) + return nil +} + +func cmdGPUShow(args []string) error { + _ = args + pref := savedGPUPreference() + if pref == nil { + fmt.Fprintln(os.Stdout, dim("No GPU preference. Set one with `runhug gpu set`.")) + return nil + } + heading(os.Stdout, "GPU preference") + printKV(os.Stdout, "provider", pref.Provider) + printKV(os.Stdout, "key", pref.Key) + if pref.Name != "" { + printKV(os.Stdout, "name", pref.Name) + } + if pref.MemoryGB > 0 { + printKV(os.Stdout, "vram", fmt.Sprintf("%.0f GB", pref.MemoryGB)) + } + return nil +} diff --git a/internal/cli/gpu_catalog.go b/internal/cli/gpu_catalog.go new file mode 100644 index 0000000..2f8655c --- /dev/null +++ b/internal/cli/gpu_catalog.go @@ -0,0 +1,523 @@ +package cli + +import ( + "context" + "fmt" + "sort" + "strings" + "time" + + "github.com/adamsiwiec1/runhug/internal/config" + "github.com/adamsiwiec1/runhug/internal/gpudb" + "github.com/adamsiwiec1/runhug/internal/hostgpu" + "github.com/adamsiwiec1/runhug/internal/runpod" +) + +// gpuRow is one display/selection row for `runhug gpu list|set`. +type gpuRow struct { + Name string `json:"name"` + Provider string `json:"provider,omitempty"` // local | runpod | gcp | "" (index-only) + Key string `json:"key,omitempty"` + MemoryGB float64 `json:"memory_gb"` + FP16 float64 `json:"fp16,omitempty"` + TensorCores int `json:"tensor_cores,omitempty"` + CUDA string `json:"cuda,omitempty"` + Arch string `json:"architecture,omitempty"` + PricePerHr float64 `json:"price_per_hour,omitempty"` + Stock string `json:"stock,omitempty"` + InStock bool `json:"in_stock,omitempty"` + Yours bool `json:"yours,omitempty"` // matches a GPU on this machine +} + +func buildGPURows(filter, query string) ([]gpuRow, error) { + filter = strings.ToLower(strings.TrimSpace(filter)) + if filter == "" { + filter = "all" + } + specs, err := gpudb.Load() + if err != nil { + return nil, err + } + if q := strings.TrimSpace(query); q != "" { + specs = gpudb.Search(specs, q) + } + + switch filter { + case "all": + return markYoursInRows(enrichIndexRows(specs)), nil + case "local": + return markYoursInRows(localCatalogRows(specs)), nil + case "amd", "nvidia": + specs = filterSpecsByVendor(specs, filter) + return enrichIndexRows(specs), nil + case "runpod": + return runpodGPURows(specs, query) + case "gcp": + return gcpGPURows(specs, query) + default: + return nil, fmt.Errorf("unknown --filter %q (want all|local|amd|nvidia|runpod|gcp)", filter) + } +} + +func filterSpecsByVendor(specs []gpudb.Spec, vendor string) []gpudb.Spec { + vendor = strings.ToLower(vendor) + out := make([]gpudb.Spec, 0, len(specs)) + for _, s := range specs { + v := strings.ToLower(s.Vendor) + if v == "" { + v = "nvidia" // legacy rows + } + if v == vendor { + out = append(out, s) + } + } + return out +} + +func enrichIndexRows(specs []gpudb.Spec) []gpuRow { + rpByName, gcpByName := cloudJoinMaps() + out := make([]gpuRow, 0, len(specs)) + for _, s := range specs { + r := rowFromSpec(s) + key := normalizeJoinKey(s.Name) + if p, ok := rpByName[key]; ok { + r.Provider = "runpod" + r.Key = p.ID + r.PricePerHr = p.PricePerHour + r.Stock = p.Availability + r.InStock = p.InStock + if p.MemoryGB > 0 { + r.MemoryGB = p.MemoryGB + } + } else if g, ok := gcpByName[key]; ok { + r.Provider = "gcp" + r.Key = g.Name + r.PricePerHr = g.SpotHourlyApprox + r.Stock = g.Series + r.InStock = true + if g.MemoryGB > 0 { + r.MemoryGB = g.MemoryGB + } + } + out = append(out, r) + } + return out +} + +func rowFromSpec(s gpudb.Spec) gpuRow { + return gpuRow{ + Name: s.Name, + MemoryGB: s.MemorySize, + FP16: s.FP16, + TensorCores: s.TensorCores, + CUDA: s.CUDA, + Arch: s.Architecture, + } +} + +// localCatalogRows is the full vendored index tagged as local-runnable SKUs +// (not "what's plugged into this machine"). Detected hardware is separate. +func localCatalogRows(specs []gpudb.Spec) []gpuRow { + devs, _ := hostgpu.Detect() + out := make([]gpuRow, 0, len(specs)) + for _, s := range specs { + r := rowFromSpec(s) + r.Provider = "local" + r.Key = s.Name + for _, d := range devs { + if d.Vendor == "apple" { + continue + } + if ms, ok := gpudb.Match(specs, d.Name); ok && normalizeJoinKey(ms.Name) == normalizeJoinKey(s.Name) { + r.Yours = true + r.Stock = "yours" + break + } + } + out = append(out, r) + } + return out +} + +// detectedLocalRows returns GPUs actually present on this host (nvidia-smi / Apple Silicon). +func detectedLocalRows(specs []gpudb.Spec) []gpuRow { + devs, err := hostgpu.Detect() + if err != nil || len(devs) == 0 { + return nil + } + out := make([]gpuRow, 0, len(devs)) + for _, d := range devs { + r := gpuRow{ + Name: d.Name, + Provider: "local", + Key: d.Name, + MemoryGB: d.MemoryGB, + FP16: d.FP16, + Yours: true, + Stock: "yours", + } + if d.Vendor == "apple" { + r.Arch = "Apple Silicon" + if d.Note != "" { + r.Arch = d.Name // keep chip name; arch column already Apple-ish via name + } + out = append(out, r) + continue + } + if s, ok := gpudb.Match(specs, d.Name); ok { + r.Name = s.Name + r.Key = s.Name + if s.FP16 > 0 { + r.FP16 = s.FP16 + } + r.TensorCores = s.TensorCores + r.CUDA = s.CUDA + r.Arch = s.Architecture + if s.MemorySize > 0 && d.MemoryGB <= 0 { + r.MemoryGB = s.MemorySize + } + } + out = append(out, r) + } + return out +} + +// markYoursInRows flags index/catalog rows that match host GPUs and injects +// Apple Silicon (or unmatched NVIDIA) so --sort best can rank them in the full list. +func markYoursInRows(rows []gpuRow) []gpuRow { + devs, err := hostgpu.Detect() + if err != nil || len(devs) == 0 { + return rows + } + specs, _ := gpudb.Load() + for _, d := range devs { + matched := false + if d.Vendor != "apple" { + for i := range rows { + if rows[i].Yours { + continue + } + nameMatch := normalizeJoinKey(rows[i].Name) == normalizeJoinKey(d.Name) + if !nameMatch && len(specs) > 0 { + if ms, ok := gpudb.Match(specs, d.Name); ok { + nameMatch = normalizeJoinKey(rows[i].Name) == normalizeJoinKey(ms.Name) + } + } + if nameMatch { + rows[i].Yours = true + rows[i].Stock = "yours" + matched = true + } + } + } + if matched { + continue + } + // Inject synthetic / unmatched host GPU into the ranking list. + inj := gpuRow{ + Name: d.Name, + Provider: "local", + Key: d.Name, + MemoryGB: d.MemoryGB, + FP16: d.FP16, + Yours: true, + Stock: "yours", + } + if d.Vendor == "apple" { + inj.Arch = "Apple Silicon" + } else if len(specs) > 0 { + if s, ok := gpudb.Match(specs, d.Name); ok { + inj.Name = s.Name + inj.Key = s.Name + inj.FP16 = s.FP16 + inj.TensorCores = s.TensorCores + inj.CUDA = s.CUDA + inj.Arch = s.Architecture + } + } + // Avoid duplicate inject if already present by key. + dup := false + for _, r := range rows { + if r.Yours && normalizeJoinKey(r.Name) == normalizeJoinKey(inj.Name) { + dup = true + break + } + } + if !dup { + rows = append(rows, inj) + } + } + return rows +} + +// yoursRank returns 1-based rank among rows already sorted, or 0 if none marked yours. +func yoursRank(rows []gpuRow) (rank int, total int, row gpuRow) { + total = len(rows) + for i, r := range rows { + if r.Yours { + return i + 1, total, r + } + } + return 0, total, gpuRow{} +} + +type localListPayload struct { + OnThisMachine []gpuRow `json:"on_this_machine"` + Catalog []gpuRow `json:"catalog"` + YoursRank int `json:"yours_rank,omitempty"` + YoursTotal int `json:"yours_total,omitempty"` +} + +func runpodGPURows(specs []gpudb.Spec, query string) ([]gpuRow, error) { + env := config.Load() + var gpus []runpod.GPU + if env.RunpodAPIKey != "" { + ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + list, err := runpod.New(env.RunpodAPIKey).ListGPUs(ctx) + if err != nil { + return nil, err + } + gpus = list + } else { + gpus = runpod.OfflineCatalog() + } + pools := runpod.SummarizePools(gpus) + q := strings.ToLower(strings.TrimSpace(query)) + out := make([]gpuRow, 0, len(pools)) + for _, p := range pools { + if q != "" { + hay := strings.ToLower(p.ID + " " + p.ExampleGPU) + if !strings.Contains(hay, q) { + continue + } + } + r := gpuRow{ + Name: p.ExampleGPU, + Provider: "runpod", + Key: p.ID, + MemoryGB: p.MemoryGB, + PricePerHr: p.PricePerHour, + Stock: p.Availability, + InStock: p.InStock, + } + if s, ok := matchCloudName(specs, p.ExampleGPU, p.ID); ok { + r.Name = s.Name + r.FP16 = s.FP16 + r.TensorCores = s.TensorCores + r.CUDA = s.CUDA + r.Arch = s.Architecture + } + out = append(out, r) + } + return out, nil +} + +func gcpGPURows(specs []gpudb.Spec, query string) ([]gpuRow, error) { + accels, err := gpudb.LoadGCP() + if err != nil { + return nil, err + } + q := strings.ToLower(strings.TrimSpace(query)) + out := make([]gpuRow, 0, len(accels)) + for _, a := range accels { + hf := a.HFName + if hf == "" { + hf = a.Name + } + if q != "" { + hay := strings.ToLower(a.ID + " " + a.Name + " " + hf + " " + a.Series) + if !strings.Contains(hay, q) { + continue + } + } + r := gpuRow{ + Name: hf, + Provider: "gcp", + Key: a.Name, + MemoryGB: a.MemoryGB, + PricePerHr: a.SpotHourlyApprox, + Stock: a.Series, + InStock: true, + } + if s, ok := gpudb.Match(specs, hf); ok { + r.Name = s.Name + if r.MemoryGB <= 0 { + r.MemoryGB = s.MemorySize + } + r.FP16 = s.FP16 + r.TensorCores = s.TensorCores + r.CUDA = s.CUDA + r.Arch = s.Architecture + } + out = append(out, r) + } + return out, nil +} + +func cloudJoinMaps() (map[string]runpod.Pool, map[string]gpudb.GCPAccelerator) { + rp := map[string]runpod.Pool{} + env := config.Load() + var gpus []runpod.GPU + if env.RunpodAPIKey != "" { + ctx, cancel := context.WithTimeout(context.Background(), 20*time.Second) + defer cancel() + if list, err := runpod.New(env.RunpodAPIKey).ListGPUs(ctx); err == nil { + gpus = list + } + } + if len(gpus) == 0 { + gpus = runpod.OfflineCatalog() + } + for _, p := range runpod.SummarizePools(gpus) { + rp[normalizeJoinKey(p.ExampleGPU)] = p + rp[normalizeJoinKey(p.ID)] = p + if s, ok := gpudb.ByName(p.ExampleGPU); ok { + rp[normalizeJoinKey(s.Name)] = p + } + } + gcpMap := map[string]gpudb.GCPAccelerator{} + if accels, err := gpudb.LoadGCP(); err == nil { + for _, a := range accels { + gcpMap[normalizeJoinKey(a.Name)] = a + if a.HFName != "" { + gcpMap[normalizeJoinKey(a.HFName)] = a + } + gcpMap[normalizeJoinKey(a.ID)] = a + } + } + return rp, gcpMap +} + +func normalizeJoinKey(s string) string { + s = strings.ToLower(strings.TrimSpace(s)) + repl := strings.NewReplacer("nvidia ", "", "geforce ", "", " ", "", "-", "", "_", "") + return repl.Replace(s) +} + +func matchCloudName(specs []gpudb.Spec, example, poolID string) (gpudb.Spec, bool) { + if s, ok := gpudb.Match(specs, example); ok { + return s, true + } + // Pool id hints: ADA_24 → 4090-ish, HOPPER_80 → H100, etc. + aliases := map[string]string{ + "ADA_24": "GeForce RTX 4090", + "ADA_48": "L40", + "AMPERE_16": "RTX A4000", + "AMPERE_48": "A40", + "AMPERE_80": "A100 SXM4 80 GB", + "HOPPER_80": "H100 SXM5 80 GB", + } + if alias, ok := aliases[strings.ToUpper(poolID)]; ok { + return gpudb.Match(specs, alias) + } + return gpudb.Spec{}, false +} + +func sortGPURows(rows []gpuRow, sortKey string) { + sortKey = strings.ToLower(strings.TrimSpace(sortKey)) + if sortKey == "" { + sortKey = "best" + } + sort.SliceStable(rows, func(i, j int) bool { + a, b := rows[i], rows[j] + switch sortKey { + case "cheapest": + ap, bp := a.PricePerHr, b.PricePerHr + if ap <= 0 && bp <= 0 { + return a.Name < b.Name + } + if ap <= 0 { + return false + } + if bp <= 0 { + return true + } + if ap != bp { + return ap < bp + } + return a.Name < b.Name + case "value": + av, bv := valueScore(a), valueScore(b) + if av != bv { + return av > bv + } + return a.Name < b.Name + case "vram": + if a.MemoryGB != b.MemoryGB { + return a.MemoryGB > b.MemoryGB + } + return a.Name < b.Name + case "name": + return strings.ToLower(a.Name) < strings.ToLower(b.Name) + default: // best + if a.FP16 != b.FP16 { + return a.FP16 > b.FP16 + } + if a.MemoryGB != b.MemoryGB { + return a.MemoryGB > b.MemoryGB + } + if a.TensorCores != b.TensorCores { + return a.TensorCores > b.TensorCores + } + return a.Name < b.Name + } + }) +} + +func valueScore(r gpuRow) float64 { + if r.PricePerHr <= 0 || r.FP16 <= 0 { + return 0 + } + return r.FP16 / r.PricePerHr +} + +func resolveGPUPreference(filter, want string) (gpuRow, error) { + rows, err := buildGPURows(filter, "") + if err != nil { + return gpuRow{}, err + } + sortGPURows(rows, "best") + want = strings.TrimSpace(want) + if want == "" { + return gpuRow{}, fmt.Errorf("empty GPU name") + } + wl := strings.ToLower(want) + for _, r := range rows { + if strings.EqualFold(r.Key, want) || strings.EqualFold(r.Name, want) { + return r, nil + } + } + for _, r := range rows { + if strings.Contains(strings.ToLower(r.Name), wl) || strings.Contains(strings.ToLower(r.Key), wl) { + return r, nil + } + } + return gpuRow{}, fmt.Errorf("no GPU matching %q under --filter %s (try `runhug gpu list`)", want, filter) +} + +func savedGPUPreference() *config.GPUPreference { + return config.LoadSettings().GPU +} + +func applySavedRunpodGPU(flagGPU string) string { + if strings.TrimSpace(flagGPU) != "" { + return flagGPU + } + p := savedGPUPreference() + if p == nil || !strings.EqualFold(p.Provider, "runpod") || p.Key == "" { + return "" + } + return p.Key +} + +func applySavedGCPGPU(flagGPU string) string { + if strings.TrimSpace(flagGPU) != "" { + return flagGPU + } + p := savedGPUPreference() + if p == nil || !strings.EqualFold(p.Provider, "gcp") || p.Key == "" { + return "" + } + return p.Key +} diff --git a/internal/cli/gpu_catalog_test.go b/internal/cli/gpu_catalog_test.go new file mode 100644 index 0000000..afe17fa --- /dev/null +++ b/internal/cli/gpu_catalog_test.go @@ -0,0 +1,54 @@ +package cli + +import ( + "testing" + + "github.com/adamsiwiec1/runhug/internal/gpudb" +) + +func TestSortGPURowsBest(t *testing.T) { + rows := []gpuRow{ + {Name: "slow", FP16: 10, MemoryGB: 24}, + {Name: "fast", FP16: 90, MemoryGB: 24}, + {Name: "mid", FP16: 40, MemoryGB: 48}, + } + sortGPURows(rows, "best") + if rows[0].Name != "fast" { + t.Fatalf("got %s", rows[0].Name) + } +} + +func TestSortGPURowsCheapest(t *testing.T) { + rows := []gpuRow{ + {Name: "a", PricePerHr: 1.5}, + {Name: "b", PricePerHr: 0.4}, + {Name: "c"}, // no price — sinks + } + sortGPURows(rows, "cheapest") + if rows[0].Name != "b" { + t.Fatalf("got %s", rows[0].Name) + } + if rows[2].Name != "c" { + t.Fatalf("unpriced should sink, got %s", rows[2].Name) + } +} + +func TestMatchCloudAlias(t *testing.T) { + specs, err := gpudb.Load() + if err != nil { + t.Fatal(err) + } + s, ok := matchCloudName(specs, "RTX 4090", "ADA_24") + if !ok || !containsFoldName(s.Name, "4090") { + t.Fatalf("ADA_24 -> %+v ok=%v", s, ok) + } +} + +func containsFoldName(s, sub string) bool { + return len(s) > 0 && (s == sub || len(sub) == 0 || + len(s) >= len(sub) && (stringContainsCI(s, sub))) +} + +func stringContainsCI(s, sub string) bool { + return len(gpudb.Search([]gpudb.Spec{{Name: s}}, sub)) == 1 +} diff --git a/internal/cli/list.go b/internal/cli/list.go index 5840343..3616940 100644 --- a/internal/cli/list.go +++ b/internal/cli/list.go @@ -504,52 +504,8 @@ func cmdImport(args []string) error { } func cmdGPUs(args []string) error { - fs := newFlagSet("gpus") - minVRAM := fs.Float64("min-vram", 0, "only pools with at least this many GB") - asJSON := fs.Bool("json", false, "print JSON") - if err := parseFlags(fs, args); err != nil { - return err - } - env := config.Load() - if err := env.RequireRunpod(); err != nil { - return err - } - ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) - defer cancel() - gpus, err := runpod.New(env.RunpodAPIKey).ListGPUs(ctx) - if err != nil { - return err - } - pools := runpod.SummarizePools(gpus) - if *asJSON { - return writeJSON(pools) - } - heading(os.Stdout, "GPUs") - fmt.Fprintf(os.Stdout, " %s %s %s %s %s\n", - dim(padRight("POOL", 12)), - dim(padRight("VRAM", 6)), - dim(padRight("$/HR", 6)), - dim(padRight("STOCK", 10)), - dim("EXAMPLE"), - ) - for _, p := range pools { - if p.MemoryGB < *minVRAM { - continue - } - stock := p.Availability - stockOut := green(padRight(stock, 10)) - if !p.InStock { - stockOut = yellow(padRight("NONE", 10)) - } - fmt.Fprintf(os.Stdout, " %s %s %s %s %s\n", - bold(padRight(p.ID, 12)), - padRight(fmt.Sprintf("%.0f", p.MemoryGB), 6), - padRight(fmt.Sprintf("%.2f", p.PricePerHour), 6), - stockOut, - dim(p.ExampleGPU), - ) - } - return nil + // Deprecated alias — kept for any external callers; routes to gpu list. + return cmdGPUList(args) } func itoaPtr(p *int) string { diff --git a/internal/cli/local.go b/internal/cli/local.go index 2a4168d..81d7af0 100644 --- a/internal/cli/local.go +++ b/internal/cli/local.go @@ -5,6 +5,7 @@ import ( "fmt" "os" "path/filepath" + "strconv" "strings" "time" @@ -27,10 +28,12 @@ func cmdLocal(args []string) error { return cmdLocalStart(args[1:]) case "stop": return cmdLocalStop(args[1:]) + case "run": + return cmdLocalRun(args[1:]) case "setup", "doctor": return cmdLocalSetup(args[1:]) default: - return fmt.Errorf("unknown local command %q (setup, add, start, stop)", args[0]) + return fmt.Errorf("unknown local command %q (setup, add, start, stop, run)", args[0]) } } @@ -366,3 +369,52 @@ func cmdLocalStop(args []string) error { fmt.Printf("stopped pid for %s\n", m.HFRepo) return nil } + +// cmdLocalRun ensures the registry model is serving, then opens the chat REPL +// (same path as `runhug run`). +func cmdLocalRun(args []string) error { + fs := newFlagSet("local run") + model := fs.String("model", "", "registry model (default: current / positional)") + port := fs.Int("port", 8081, "llama-server port when starting") + ctxSize := fs.Int("ctx", 4096, "context tokens (llama-server -c)") + threads := fs.Int("threads", 0, "CPU threads (0 = all)") + bin := fs.String("bin", "", "llama-server binary") + oneshot := fs.String("q", "", "one-shot prompt then exit") + stream := fs.Bool("stream", true, "stream chat completions") + if err := parseFlags(fs, args); err != nil { + return err + } + key := strings.TrimSpace(*model) + if key == "" && fs.NArg() > 0 { + key = fs.Arg(0) + } + reg, _, err := store.Load() + if err != nil { + return err + } + m, ok := reg.Lookup(key) + if !ok { + return fmt.Errorf("unknown local model %q — `runhug local add` first", key) + } + needStart := m.BaseURL == "" || (m.Runtime != runtime.Ollama && m.Runtime != runtime.MLX && m.LocalPID == 0 && m.GGUFPath != "") + if needStart { + startArgs := []string{"--model", m.HFRepo, "--port", strconv.Itoa(*port), "--ctx", strconv.Itoa(*ctxSize)} + if *threads > 0 { + startArgs = append(startArgs, "--threads", strconv.Itoa(*threads)) + } + if *bin != "" { + startArgs = append(startArgs, "--bin", *bin) + } + if err := cmdLocalStart(startArgs); err != nil { + return err + } + } + runArgs := []string{"--yes", m.HFRepo} + if q := strings.TrimSpace(*oneshot); q != "" { + runArgs = append(runArgs, "-q", q) + } + if !*stream { + runArgs = append(runArgs, "--stream=false") + } + return cmdRun(runArgs) +} diff --git a/internal/cli/recommend.go b/internal/cli/recommend.go index aa61710..54d2e4a 100644 --- a/internal/cli/recommend.go +++ b/internal/cli/recommend.go @@ -243,7 +243,11 @@ func cmdRecommendGPU(args []string) error { } live := loadGPUCatalog(ctx) - adv, err := recommend.AdviseGPU(*model, live, *gpuPool, *maxLen) + prefer := applySavedRunpodGPU(*gpuPool) + if prefer != "" && strings.TrimSpace(*gpuPool) == "" { + fmt.Fprintf(os.Stderr, "%s using saved GPU preference %s\n", dim("note:"), prefer) + } + adv, err := recommend.AdviseGPU(*model, live, prefer, *maxLen) if err != nil { return err } diff --git a/internal/cli/root.go b/internal/cli/root.go index e3b53be..4220f4a 100644 --- a/internal/cli/root.go +++ b/internal/cli/root.go @@ -72,8 +72,10 @@ func Run(args []string) error { return cmdDelete(rest) case "status": return cmdStatus(rest) + case "gpu": + return cmdGPU(rest) case "gpus": - return cmdGPUs(rest) + return cmdGPUList(rest) case "import": return cmdImport(rest) case "version", "-v", "--version": @@ -122,8 +124,13 @@ func printUsage(w io.Writer) { fmt.Fprintln(w, " start opencode wire opencode to the proxy/deployment") fmt.Fprintln(w) fmt.Fprintln(w, bold("local")) - fmt.Fprintln(w, " local add models on this machine") - fmt.Fprintln(w, " local setup set up local runtimes") + fmt.Fprintln(w, " local add|start|stop|run|setup models on this machine") + fmt.Fprintln(w) + fmt.Fprintln(w, bold("gpu")) + fmt.Fprintln(w, " gpu list hardware index + runpod/gcp/local") + fmt.Fprintln(w, " gpu set|clear|show preference for deploy / local") + fmt.Fprintln(w, " gpu update refresh NVIDIA + GCP catalogs") + fmt.Fprintln(w, " gpus alias for gpu list") fmt.Fprintln(w) fmt.Fprintln(w, bold("config")) fmt.Fprintln(w, " config config dir + settings") diff --git a/internal/cli/setup.go b/internal/cli/setup.go index 820d748..da13e07 100644 --- a/internal/cli/setup.go +++ b/internal/cli/setup.go @@ -4,6 +4,7 @@ import ( "fmt" "os" + "github.com/adamsiwiec1/runhug/internal/hostgpu" "github.com/adamsiwiec1/runhug/internal/runtime" ) @@ -17,6 +18,7 @@ func cmdLocalSetup(args []string) error { eng, err := resolveEngine(*want, *yes, true) if err != nil { runtime.PrintConfig(os.Stderr, runtime.Detect()) + printLocalGPUs(os.Stderr) return err } fmt.Printf("%s %s %s\n", dim("using"), eng.Kind, dim(dash(eng.Binary))) @@ -27,9 +29,21 @@ func cmdLocalSetup(args []string) error { } fmt.Fprintln(os.Stderr) runtime.PrintConfig(os.Stderr, runtime.Detect()) + printLocalGPUs(os.Stderr) return nil } +func printLocalGPUs(w *os.File) { + devs, err := hostgpu.Detect() + if err != nil || len(devs) == 0 { + fmt.Fprintln(w, dim("gpu (none detected — runhug gpu list --filter local)")) + return + } + for _, d := range devs { + fmt.Fprintf(w, "%s %s ~%.0f GB (%s)\n", dim("gpu"), d.Name, d.MemoryGB, d.Source) + } +} + func resolveEngine(want string, yes, printMissing bool) (runtime.Engine, error) { want = runtime.Normalize(want) snap := runtime.Detect() diff --git a/internal/cli/wizard.go b/internal/cli/wizard.go index 6d8a5bb..1e93bc0 100644 --- a/internal/cli/wizard.go +++ b/internal/cli/wizard.go @@ -451,7 +451,8 @@ func wizardGPU(w io.Writer, modelID string) (string, error) { return "", err } live := loadGPUCatalog(ctx) - adv, err := recommend.AdviseGPU(*model, live, "", 8192) + prefer := applySavedRunpodGPU("") + adv, err := recommend.AdviseGPU(*model, live, prefer, 8192) if err != nil { return "", err } diff --git a/internal/config/config.go b/internal/config/config.go index 42088d1..7a7ea87 100644 --- a/internal/config/config.go +++ b/internal/config/config.go @@ -296,12 +296,21 @@ func loadStoredHFToken() string { return SanitizeAPIKey(string(raw)) } +// GPUPreference is the user's saved default accelerator for deploy / local flows. +type GPUPreference struct { + Provider string `json:"provider"` // local | runpod | gcp + Key string `json:"key"` // pool id, L4/T4, or local matched name + Name string `json:"name,omitempty"` + MemoryGB float64 `json:"memory_gb,omitempty"` +} + // Settings holds optional CLI defaults under ~/.config/runhug/settings.json. type Settings struct { - NoColor bool `json:"no_color"` - UpdateLimit *int `json:"update_limit,omitempty"` // Hub update row cap; 0=unlimited; nil=default - AdvisorBaseURL string `json:"advisor_base_url,omitempty"` - AdvisorModel string `json:"advisor_model,omitempty"` + NoColor bool `json:"no_color"` + UpdateLimit *int `json:"update_limit,omitempty"` // Hub update row cap; 0=unlimited; nil=default + AdvisorBaseURL string `json:"advisor_base_url,omitempty"` + AdvisorModel string `json:"advisor_model,omitempty"` + GPU *GPUPreference `json:"gpu,omitempty"` } func SettingsPath() (string, error) { diff --git a/internal/gpudb/cache.go b/internal/gpudb/cache.go new file mode 100644 index 0000000..6aabe65 --- /dev/null +++ b/internal/gpudb/cache.go @@ -0,0 +1,224 @@ +package gpudb + +import ( + _ "embed" + "encoding/json" + "fmt" + "os" + "path/filepath" + "strings" + "sync" + + "github.com/adamsiwiec1/runhug/internal/config" +) + +//go:embed data/gcp.json +var gcpJSON []byte + +// GCPAccelerator is one Compute Engine GPU offering (accelerator type or A/G-series chip). +type GCPAccelerator struct { + ID string `json:"id"` // e.g. nvidia-tesla-t4, nvidia-l4 + Name string `json:"name"` + HFName string `json:"hfName,omitempty"` // gpudb Spec name for join + Series string `json:"series,omitempty"` // N1 | G2 | A2 | A3 | A4 | G4 + MemoryGB float64 `json:"memoryGB"` + MachineHint string `json:"machineHint,omitempty"` + Attach string `json:"attach,omitempty"` // accelerator | machine + SpotHourlyApprox float64 `json:"spotHourlyApprox,omitempty"` +} + +const ( + cacheSubdir = "gpudb" + nvidiaCache = "nvidia.json" + amdCache = "amd.json" + gcpCache = "gcp.json" + NVIDIAFetchURL = "https://huggingface.co/datasets/Jr23xd23/gpu-database/resolve/main/data/nvidia/all.json" + AMDFetchURL = "https://huggingface.co/datasets/Jr23xd23/gpu-database/resolve/main/data/amd/all.json" + // GCPCatalogURL is the shipped catalog on the public repo (same path as embed). + GCPCatalogURL = "https://raw.githubusercontent.com/openhat-security/runhug/main/internal/gpudb/data/gcp.json" +) + +var ( + loadOnce sync.Once + allSpecs []Spec + loadErr error + + gcpOnce sync.Once + gcpAll []GCPAccelerator + gcpErr error + specMu sync.Mutex +) + +// CacheDir is ~/.config/runhug/gpudb (or RUNHUG_CONFIG parent). +func CacheDir() (string, error) { + dir, err := config.Dir() + if err != nil { + return "", err + } + return filepath.Join(dir, cacheSubdir), nil +} + +// Invalidate clears in-memory caches so the next Load* re-reads embed/disk. +func Invalidate() { + specMu.Lock() + defer specMu.Unlock() + loadOnce = sync.Once{} + allSpecs = nil + loadErr = nil + gcpOnce = sync.Once{} + gcpAll = nil + gcpErr = nil +} + +// Load returns NVIDIA + AMD specs, preferring updated caches over embeds. +func Load() ([]Spec, error) { + specMu.Lock() + defer specMu.Unlock() + loadOnce.Do(func() { + nv, err := loadVendorSpecs(nvidiaCache, nvidiaJSON, "nvidia") + if err != nil { + loadErr = err + return + } + amd, err := loadVendorSpecs(amdCache, amdJSON, "amd") + if err != nil { + loadErr = err + return + } + allSpecs = append(nv, amd...) + }) + if loadErr != nil { + return nil, loadErr + } + out := make([]Spec, len(allSpecs)) + copy(out, allSpecs) + return out, nil +} + +func loadVendorSpecs(cacheName string, embed []byte, vendor string) ([]Spec, error) { + var specs []Spec + if raw, err := readCache(cacheName); err == nil && len(raw) > 0 { + if err := json.Unmarshal(raw, &specs); err != nil { + return nil, fmt.Errorf("gpudb: decode cached %s: %w", cacheName, err) + } + } else { + if err := json.Unmarshal(embed, &specs); err != nil { + return nil, fmt.Errorf("gpudb: decode %s: %w", cacheName, err) + } + } + for i := range specs { + if specs[i].Vendor == "" { + specs[i].Vendor = vendor + } + } + return specs, nil +} + +// LoadGCP returns GCP accelerator catalog (cache over embed). +func LoadGCP() ([]GCPAccelerator, error) { + specMu.Lock() + defer specMu.Unlock() + gcpOnce.Do(func() { + if raw, err := readCache(gcpCache); err == nil && len(raw) > 0 { + if err := json.Unmarshal(raw, &gcpAll); err != nil { + gcpErr = fmt.Errorf("gpudb: decode cached gcp.json: %w", err) + return + } + return + } + if err := json.Unmarshal(gcpJSON, &gcpAll); err != nil { + gcpErr = fmt.Errorf("gpudb: decode gcp.json: %w", err) + return + } + }) + if gcpErr != nil { + return nil, gcpErr + } + out := make([]GCPAccelerator, len(gcpAll)) + copy(out, gcpAll) + return out, nil +} + +// SourceLabel reports whether data is from cache or the shipped embed. +func SourceLabel(kind string) string { + name := nvidiaCache + switch kind { + case "gcp": + name = gcpCache + case "amd": + name = amdCache + } + if _, err := readCache(name); err == nil { + return "cache (~/.config/runhug/gpudb)" + } + return "bundled" +} + +func readCache(name string) ([]byte, error) { + dir, err := CacheDir() + if err != nil { + return nil, err + } + return os.ReadFile(filepath.Join(dir, name)) +} + +// WriteCache writes a catalog file under CacheDir and invalidates memory. +func WriteCache(name string, raw []byte) error { + dir, err := CacheDir() + if err != nil { + return err + } + if err := os.MkdirAll(dir, 0o700); err != nil { + return err + } + path := filepath.Join(dir, name) + if err := os.WriteFile(path, raw, 0o600); err != nil { + return err + } + Invalidate() + return nil +} + +// TrimNVIDIA filters full TechPowerUp dump to CUDA≥7.5 VRAM≥8GB compact rows. +func TrimNVIDIA(full []Spec) []Spec { + out := make([]Spec, 0, len(full)) + for _, g := range full { + if g.MemorySize < 8 { + continue + } + cuda := 0.0 + if g.CUDA != "" { + fmt.Sscanf(g.CUDA, "%f", &cuda) + } + if cuda < 7.5 { + continue + } + g.Vendor = "nvidia" + out = append(out, g) + } + return out +} + +// TrimAMD keeps VRAM≥8GB Radeon / Instinct (and rows with FP16/FP32). +func TrimAMD(full []Spec) []Spec { + out := make([]Spec, 0, len(full)) + for _, g := range full { + if g.MemorySize < 8 { + continue + } + low := strings.ToLower(g.Name) + keep := g.FP16 > 0 || g.FP32 > 0 + for _, p := range []string{"instinct", "radeon rx", "radeon pro", "radeon vii", "firepro"} { + if strings.Contains(low, p) { + keep = true + break + } + } + if !keep { + continue + } + g.Vendor = "amd" + out = append(out, g) + } + return out +} diff --git a/internal/gpudb/data/amd.json b/internal/gpudb/data/amd.json new file mode 100644 index 0000000..9d05685 --- /dev/null +++ b/internal/gpudb/data/amd.json @@ -0,0 +1 @@ +[{"name":"Radeon Instinct MI325X","vendor":"amd","architecture":"CDNA 3.0","generation":"Radeon Instinct(MIx)","memorySize":288.0,"memoryBandwidth":10300.0,"memoryType":"HBM3e","fp16":653.7,"fp32":81.72,"fp64":81.72,"shaders":19456,"tdp":1000},{"name":"Radeon Instinct MI300A","vendor":"amd","architecture":"CDNA 3.0","generation":"Radeon Instinct(MIx)","memorySize":192.0,"memoryBandwidth":10300.0,"memoryType":"HBM3","fp16":653.7,"fp32":81.72,"fp64":81.72,"shaders":19456,"tdp":750},{"name":"Radeon Instinct MI300X","vendor":"amd","architecture":"CDNA 3.0","generation":"Radeon Instinct(MIx)","memorySize":192.0,"memoryBandwidth":10300.0,"memoryType":"HBM3","fp16":653.7,"fp32":81.72,"fp64":81.72,"shaders":19456,"tdp":750},{"name":"Radeon Instinct MI308X","vendor":"amd","architecture":"CDNA 3.0","generation":"Radeon Instinct(MIx)","memorySize":192.0,"memoryBandwidth":10300.0,"memoryType":"HBM3","fp16":653.7,"fp32":81.72,"fp64":81.72,"shaders":19456,"tdp":750},{"name":"Radeon Instinct MI355X","vendor":"amd","architecture":"CDNA 4.0","generation":"Radeon Instinct(MIx)","memorySize":288.0,"memoryBandwidth":8189.999999999999,"memoryType":"HBM3e","fp16":629.1,"fp32":78.64,"fp64":78.64,"shaders":16384,"tdp":1400},{"name":"Radeon Instinct MI350X","vendor":"amd","architecture":"CDNA 4.0","generation":"Radeon Instinct(MIx)","memorySize":288.0,"memoryBandwidth":8189.999999999999,"memoryType":"HBM3e","fp16":576.7,"fp32":72.09,"fp64":72.09,"shaders":16384,"tdp":1000},{"name":"Radeon Instinct MI250X","vendor":"amd","architecture":"CDNA 2.0","generation":"Radeon Instinct(MIx)","memorySize":128.0,"memoryBandwidth":3280.0,"memoryType":"HBM2e","fp16":383.0,"fp32":47.87,"fp64":47.87,"shaders":14080,"tdp":500},{"name":"Radeon Instinct MI300","vendor":"amd","architecture":"CDNA 3.0","generation":"Radeon Instinct(MIx)","memorySize":128.0,"memoryBandwidth":6550.0,"memoryType":"HBM3","fp16":383.0,"fp32":47.87,"fp64":47.87,"shaders":14080,"tdp":600},{"name":"Radeon Instinct MI250","vendor":"amd","architecture":"CDNA 2.0","generation":"Radeon Instinct(MIx)","memorySize":128.0,"memoryBandwidth":3280.0,"memoryType":"HBM2e","fp16":362.1,"fp32":45.26,"fp64":45.26,"shaders":13312,"tdp":500},{"name":"Radeon Instinct MI100","vendor":"amd","architecture":"CDNA 1.0","generation":"Radeon Instinct(MIx)","memorySize":32.0,"memoryBandwidth":1230.0,"memoryType":"HBM2","fp16":184.6,"fp32":23.07,"fp64":11.54,"shaders":7680,"tdp":300},{"name":"Radeon Instinct MI200","vendor":"amd","architecture":"CDNA 2.0","generation":"Radeon Instinct(MIx)","memorySize":64.0,"memoryBandwidth":1640.0,"memoryType":"HBM2e","fp16":181.0,"fp32":22.63,"fp64":22.63,"shaders":6656,"tdp":300},{"name":"Radeon Instinct MI210","vendor":"amd","architecture":"CDNA 2.0","generation":"Radeon Instinct(MIx)","memorySize":64.0,"memoryBandwidth":1640.0,"memoryType":"HBM2e","fp16":181.0,"fp32":22.63,"fp64":22.63,"shaders":6656,"tdp":300},{"name":"Radeon RX 7900 XTX","vendor":"amd","architecture":"RDNA 3.0","generation":"Navi III(RX 7000)","memorySize":24.0,"memoryBandwidth":960.0,"memoryType":"GDDR6","fp16":122.8,"fp32":61.39,"fp64":1.92,"shaders":6144,"tdp":355},{"name":"Radeon PRO W7900","vendor":"amd","architecture":"RDNA 3.0","generation":"Radeon Pro Navi(Navi III Series)","memorySize":48.0,"memoryBandwidth":864.0,"memoryType":"GDDR6","fp16":122.6,"fp32":61.32,"fp64":1.92,"shaders":6144,"tdp":295},{"name":"Radeon PRO W7900D","vendor":"amd","architecture":"RDNA 3.0","generation":"Radeon Pro Navi(Navi III Series)","memorySize":48.0,"memoryBandwidth":864.0,"memoryType":"GDDR6","fp16":106.0,"fp32":52.99,"fp64":1.66,"shaders":6144,"tdp":295},{"name":"Radeon RX 7900 XT","vendor":"amd","architecture":"RDNA 3.0","generation":"Navi III(RX 7000)","memorySize":20.0,"memoryBandwidth":800.0,"memoryType":"GDDR6","fp16":103.0,"fp32":51.48,"fp64":1.61,"shaders":5376,"tdp":300},{"name":"Radeon RX 9070 XT","vendor":"amd","architecture":"RDNA 4.0","generation":"Navi IV(RX 9000)","memorySize":16.0,"memoryBandwidth":644.6,"memoryType":"GDDR6","fp16":97.32,"fp32":48.66,"fp64":1.52,"shaders":4096,"tdp":304},{"name":"Radeon AI PRO R9700","vendor":"amd","architecture":"RDNA 4.0","generation":"Radeon Pro Navi(Navi IV Series)","memorySize":32.0,"memoryBandwidth":644.6,"memoryType":"GDDR6","fp16":95.68,"fp32":47.84,"fp64":1.5,"shaders":4096,"tdp":300},{"name":"Radeon AI PRO R9700S","vendor":"amd","architecture":"RDNA 4.0","generation":"Radeon Pro Navi(Navi IV Series)","memorySize":32.0,"memoryBandwidth":644.6,"memoryType":"GDDR6","fp16":95.68,"fp32":47.84,"fp64":1.5,"shaders":4096,"tdp":300},{"name":"Radeon RX 7900 GRE","vendor":"amd","architecture":"RDNA 3.0","generation":"Navi III(RX 7000)","memorySize":16.0,"memoryBandwidth":576.0,"memoryType":"GDDR6","fp16":91.96,"fp32":45.98,"fp64":1.44,"shaders":5120,"tdp":260},{"name":"Radeon PRO W7800 48 GB","vendor":"amd","architecture":"RDNA 3.0","generation":"Radeon Pro Navi(Navi III Series)","memorySize":48.0,"memoryBandwidth":864.0,"memoryType":"GDDR6","fp16":90.5,"fp32":45.25,"fp64":1.41,"shaders":4480,"tdp":281},{"name":"Radeon PRO W7800","vendor":"amd","architecture":"RDNA 3.0","generation":"Radeon Pro Navi(Navi III Series)","memorySize":32.0,"memoryBandwidth":576.0,"memoryType":"GDDR6","fp16":90.5,"fp32":45.25,"fp64":1.41,"shaders":4480,"tdp":260},{"name":"Radeon RX 7900M","vendor":"amd","architecture":"RDNA 3.0","generation":"Navi Mobile(RX 7000M)","memorySize":16.0,"memoryBandwidth":576.0,"memoryType":"GDDR6","fp16":77.05,"fp32":38.52,"fp64":1.2,"shaders":4608,"tdp":180},{"name":"Radeon RX 7800 XT","vendor":"amd","architecture":"RDNA 3.0","generation":"Navi III(RX 7000)","memorySize":16.0,"memoryBandwidth":624.1,"memoryType":"GDDR6","fp16":74.65,"fp32":37.32,"fp64":1.17,"shaders":3840,"tdp":263},{"name":"Radeon RX 9070","vendor":"amd","architecture":"RDNA 4.0","generation":"Navi IV(RX 9000)","memorySize":16.0,"memoryBandwidth":644.6,"memoryType":"GDDR6","fp16":72.25,"fp32":36.13,"fp64":1.13,"shaders":3584,"tdp":220},{"name":"Radeon RX 7800M","vendor":"amd","architecture":"RDNA 3.0","generation":"Navi Mobile(RX 7000M)","memorySize":12.0,"memoryBandwidth":432.0,"memoryType":"GDDR6","fp16":71.73,"fp32":35.87,"fp64":1.12,"shaders":3840,"tdp":180},{"name":"Radeon RX 7700 XT","vendor":"amd","architecture":"RDNA 3.0","generation":"Navi III(RX 7000)","memorySize":12.0,"memoryBandwidth":432.0,"memoryType":"GDDR6","fp16":70.34,"fp32":35.17,"fp64":1.1,"shaders":3456,"tdp":245},{"name":"Radeon RX 9070 GRE 16 GB","vendor":"amd","architecture":"RDNA 4.0","generation":"Navi IV(RX 9000)","memorySize":16.0,"memoryBandwidth":576.0,"memoryType":"GDDR6","fp16":68.57,"fp32":34.28,"fp64":1.07,"shaders":3072,"tdp":220},{"name":"Radeon RX 9070 GRE","vendor":"amd","architecture":"RDNA 4.0","generation":"Navi IV(RX 9000)","memorySize":12.0,"memoryBandwidth":432.0,"memoryType":"GDDR6","fp16":68.57,"fp32":34.28,"fp64":1.07,"shaders":3072,"tdp":220},{"name":"Radeon PRO W7700","vendor":"amd","architecture":"RDNA 3.0","generation":"Radeon Pro Navi(Navi III Series)","memorySize":16.0,"memoryBandwidth":576.0,"memoryType":"GDDR6","fp16":63.9,"fp32":31.95,"fp64":1.0,"shaders":3072,"tdp":190},{"name":"Radeon PRO V710","vendor":"amd","architecture":"RDNA 3.0","generation":"Radeon Pro Navi(Navi III Series)","memorySize":28.0,"memoryBandwidth":504.0,"memoryType":"GDDR6","fp16":55.3,"fp32":27.65,"fp64":0.86,"shaders":3456,"tdp":158},{"name":"Radeon RX 7700","vendor":"amd","architecture":"RDNA 3.0","generation":"Navi III(RX 7000)","memorySize":16.0,"memoryBandwidth":622.1,"memoryType":"GDDR6","fp16":53.25,"fp32":26.62,"fp64":0.83,"shaders":2560,"tdp":200},{"name":"Radeon RX 9060 XT 16 GB","vendor":"amd","architecture":"RDNA 4.0","generation":"Navi IV(RX 9000)","memorySize":16.0,"memoryBandwidth":322.3,"memoryType":"GDDR6","fp16":51.28,"fp32":25.64,"fp64":0.8,"shaders":2048,"tdp":160},{"name":"Radeon RX 9060 XT 8 GB","vendor":"amd","architecture":"RDNA 4.0","generation":"Navi IV(RX 9000)","memorySize":8.0,"memoryBandwidth":322.3,"memoryType":"GDDR6","fp16":51.28,"fp32":25.64,"fp64":0.8,"shaders":2048,"tdp":150},{"name":"Radeon AI PRO 9600D","vendor":"amd","architecture":"RDNA 4.0","generation":"Radeon Pro Navi(Navi IV Series)","memorySize":32.0,"memoryBandwidth":576.0,"memoryType":"GDDR6","fp16":49.64,"fp32":24.82,"fp64":0.78,"shaders":3072,"tdp":150},{"name":"Radeon RX 6950 XT","vendor":"amd","architecture":"RDNA 2.0","generation":"Navi II(RX 6000)","memorySize":16.0,"memoryBandwidth":576.0,"memoryType":"GDDR6","fp16":47.31,"fp32":23.65,"fp64":1.48,"shaders":5120,"tdp":335},{"name":"Radeon RX 6900 XT","vendor":"amd","architecture":"RDNA 2.0","generation":"Navi II(RX 6000)","memorySize":16.0,"memoryBandwidth":512.0,"memoryType":"GDDR6","fp16":46.08,"fp32":23.04,"fp64":1.44,"shaders":5120,"tdp":300},{"name":"Radeon RX 7600 XT","vendor":"amd","architecture":"RDNA 3.0","generation":"Navi III(RX 7000)","memorySize":16.0,"memoryBandwidth":288.0,"memoryType":"GDDR6","fp16":45.14,"fp32":22.57,"fp64":0.71,"shaders":2048,"tdp":190},{"name":"Radeon Pro W6900X","vendor":"amd","architecture":"RDNA 2.0","generation":"Radeon Pro Mac(Navi II Series)","memorySize":32.0,"memoryBandwidth":512.0,"memoryType":"GDDR6","fp16":44.46,"fp32":22.23,"fp64":1.39,"shaders":5120,"tdp":300},{"name":"Radeon RX 7650 GRE","vendor":"amd","architecture":"RDNA 3.0","generation":"Navi III(RX 7000)","memorySize":8.0,"memoryBandwidth":288.0,"memoryType":"GDDR6","fp16":44.15,"fp32":22.08,"fp64":0.69,"shaders":2048,"tdp":165},{"name":"Radeon RX 7600","vendor":"amd","architecture":"RDNA 3.0","generation":"Navi III(RX 7000)","memorySize":8.0,"memoryBandwidth":288.0,"memoryType":"GDDR6","fp16":43.5,"fp32":21.75,"fp64":0.68,"shaders":2048,"tdp":165},{"name":"Radeon RX 9060","vendor":"amd","architecture":"RDNA 4.0","generation":"Navi IV(RX 9000)","memorySize":8.0,"memoryBandwidth":322.3,"memoryType":"GDDR6","fp16":42.86,"fp32":21.43,"fp64":0.67,"shaders":1792,"tdp":132},{"name":"Radeon RX 6800 XT","vendor":"amd","architecture":"RDNA 2.0","generation":"Navi II(RX 6000)","memorySize":16.0,"memoryBandwidth":512.0,"memoryType":"GDDR6","fp16":41.47,"fp32":20.74,"fp64":1.3,"shaders":4608,"tdp":300},{"name":"Radeon RX 7700S","vendor":"amd","architecture":"RDNA 3.0","generation":"Navi Mobile(RX 7000M)","memorySize":8.0,"memoryBandwidth":288.0,"memoryType":"GDDR6","fp16":40.96,"fp32":20.48,"fp64":0.64,"shaders":2048,"tdp":100},{"name":"Radeon PRO V620","vendor":"amd","architecture":"RDNA 2.0","generation":"Radeon Pro Navi(Navi II Series)","memorySize":32.0,"memoryBandwidth":512.0,"memoryType":"GDDR6","fp16":40.55,"fp32":20.28,"fp64":1.27,"shaders":4608,"tdp":300},{"name":"Radeon RX 7600M XT","vendor":"amd","architecture":"RDNA 3.0","generation":"Navi Mobile(RX 7000M)","memorySize":8.0,"memoryBandwidth":288.0,"memoryType":"GDDR6","fp16":40.45,"fp32":20.23,"fp64":0.63,"shaders":2048,"tdp":120},{"name":"Radeon PRO W7600","vendor":"amd","architecture":"RDNA 3.0","generation":"Radeon Pro Navi(Navi III Series)","memorySize":8.0,"memoryBandwidth":288.0,"memoryType":"GDDR6","fp16":39.98,"fp32":19.99,"fp64":0.62,"shaders":2048,"tdp":130},{"name":"Playstation 5 Pro GPU","vendor":"amd","architecture":"RDNA 2.0","generation":"Console GPU(Sony)","memorySize":16.0,"memoryBandwidth":576.0,"memoryType":"GDDR6","fp16":36.1,"fp32":18.05,"fp64":1.13,"shaders":3840,"tdp":232},{"name":"Radeon PRO W6800","vendor":"amd","architecture":"RDNA 2.0","generation":"Radeon Pro Navi(Navi II Series)","memorySize":32.0,"memoryBandwidth":512.0,"memoryType":"GDDR6","fp16":35.67,"fp32":17.83,"fp64":1.11,"shaders":3840,"tdp":250},{"name":"Radeon RX 7600M","vendor":"amd","architecture":"RDNA 3.0","generation":"Navi Mobile(RX 7000M)","memorySize":8.0,"memoryBandwidth":256.0,"memoryType":"GDDR6","fp16":34.55,"fp32":17.27,"fp64":0.54,"shaders":1792,"tdp":90},{"name":"Radeon RX 7400","vendor":"amd","architecture":"RDNA 3.0","generation":"Navi III(RX 7000)","memorySize":8.0,"memoryBandwidth":288.0,"memoryType":"GDDR6","fp16":32.97,"fp32":16.49,"fp64":0.52,"shaders":1792,"tdp":43},{"name":"Radeon RX 6800","vendor":"amd","architecture":"RDNA 2.0","generation":"Navi II(RX 6000)","memorySize":16.0,"memoryBandwidth":512.0,"memoryType":"GDDR6","fp16":32.33,"fp32":16.17,"fp64":1.01,"shaders":3840,"tdp":250},{"name":"Radeon Pro W6800X","vendor":"amd","architecture":"RDNA 2.0","generation":"Radeon Pro Mac(Navi II Series)","memorySize":32.0,"memoryBandwidth":512.0,"memoryType":"GDDR6","fp16":32.06,"fp32":16.03,"fp64":1.0,"shaders":3840,"tdp":200},{"name":"Radeon RX 7600S","vendor":"amd","architecture":"RDNA 3.0","generation":"Navi Mobile(RX 7000M)","memorySize":8.0,"memoryBandwidth":256.0,"memoryType":"GDDR6","fp16":31.54,"fp32":15.77,"fp64":0.49,"shaders":1792,"tdp":75},{"name":"Radeon Pro W6800X Duo","vendor":"amd","architecture":"RDNA 2.0","generation":"Radeon Pro Mac(Navi II Series)","memorySize":32.0,"memoryBandwidth":512.0,"memoryType":"GDDR6","fp16":30.21,"fp32":15.11,"fp64":0.94,"shaders":3840,"tdp":400},{"name":"Radeon Instinct MI60","vendor":"amd","architecture":"GCN 5.1","generation":"Radeon Instinct(MIx)","memorySize":32.0,"memoryBandwidth":1020.0,"memoryType":"HBM2","fp16":29.49,"fp32":14.75,"fp64":7.37,"shaders":4096,"tdp":300},{"name":"Radeon Pro Vega II","vendor":"amd","architecture":"GCN 5.1","generation":"Radeon Pro Mac(Vega Series)","memorySize":32.0,"memoryBandwidth":825.3,"memoryType":"HBM2","fp16":28.18,"fp32":14.09,"fp64":7.04,"shaders":4096,"tdp":475},{"name":"Radeon Pro Vega II Duo","vendor":"amd","architecture":"GCN 5.1","generation":"Radeon Pro Mac(Vega Series)","memorySize":32.0,"memoryBandwidth":1020.0,"memoryType":"HBM2","fp16":28.18,"fp32":14.09,"fp64":7.04,"shaders":4096,"tdp":475},{"name":"Radeon RX Vega 64 Liquid Cooling","vendor":"amd","architecture":"GCN 5.0","generation":"Vega(RX Vega)","memorySize":8.0,"memoryBandwidth":483.8,"memoryType":"HBM2","fp16":27.48,"fp32":13.74,"fp64":0.86,"shaders":4096,"tdp":345},{"name":"Radeon VII","vendor":"amd","architecture":"GCN 5.1","generation":"Vega II(Radeon VII)","memorySize":16.0,"memoryBandwidth":1020.0,"memoryType":"HBM2","fp16":26.88,"fp32":13.44,"fp64":3.36,"shaders":3840,"tdp":295},{"name":"Radeon Instinct MI50","vendor":"amd","architecture":"GCN 5.1","generation":"Radeon Instinct(MIx)","memorySize":16.0,"memoryBandwidth":1020.0,"memoryType":"HBM2","fp16":26.82,"fp32":13.41,"fp64":6.71,"shaders":3840,"tdp":300},{"name":"Radeon RX 6750 XT","vendor":"amd","architecture":"RDNA 2.0","generation":"Navi II(RX 6000)","memorySize":12.0,"memoryBandwidth":432.0,"memoryType":"GDDR6","fp16":26.62,"fp32":13.31,"fp64":0.83,"shaders":2560,"tdp":250},{"name":"Radeon RX 6700 XT","vendor":"amd","architecture":"RDNA 2.0","generation":"Navi II(RX 6000)","memorySize":12.0,"memoryBandwidth":384.0,"memoryType":"GDDR6","fp16":26.43,"fp32":13.21,"fp64":0.83,"shaders":2560,"tdp":230},{"name":"Radeon RX 6750 GRE 12 GB","vendor":"amd","architecture":"RDNA 2.0","generation":"Navi II(RX 6000)","memorySize":12.0,"memoryBandwidth":432.0,"memoryType":"GDDR6","fp16":26.43,"fp32":13.21,"fp64":0.83,"shaders":2560,"tdp":250},{"name":"Radeon RX 6850M XT","vendor":"amd","architecture":"RDNA 2.0","generation":"Navi Mobile(RX 6000M)","memorySize":12.0,"memoryBandwidth":432.0,"memoryType":"GDDR6","fp16":26.43,"fp32":13.21,"fp64":0.83,"shaders":2560,"tdp":165},{"name":"Radeon Vega Frontier Edition","vendor":"amd","architecture":"GCN 5.0","generation":"Radeon Pro Vega(Vega Series)","memorySize":16.0,"memoryBandwidth":483.8,"memoryType":"HBM2","fp16":26.21,"fp32":13.11,"fp64":0.82,"shaders":4096,"tdp":300},{"name":"Radeon Vega Frontier Edition Watercooled","vendor":"amd","architecture":"GCN 5.0","generation":"Radeon Pro Vega(Vega Series)","memorySize":16.0,"memoryBandwidth":483.8,"memoryType":"HBM2","fp16":26.21,"fp32":13.11,"fp64":0.82,"shaders":4096,"tdp":375},{"name":"Radeon Pro VII","vendor":"amd","architecture":"GCN 5.1","generation":"Radeon Pro Vega(Vega II Series)","memorySize":16.0,"memoryBandwidth":1020.0,"memoryType":"HBM2","fp16":26.11,"fp32":13.06,"fp64":6.53,"shaders":3840,"tdp":250},{"name":"Radeon RX Vega 64","vendor":"amd","architecture":"GCN 5.0","generation":"Vega(RX Vega)","memorySize":8.0,"memoryBandwidth":483.8,"memoryType":"HBM2","fp16":25.33,"fp32":12.66,"fp64":0.79,"shaders":4096,"tdp":295},{"name":"Radeon RX Vega 64 Limited Edition","vendor":"amd","architecture":"GCN 5.0","generation":"Vega(RX Vega)","memorySize":8.0,"memoryBandwidth":483.8,"memoryType":"HBM2","fp16":25.33,"fp32":12.66,"fp64":0.79,"shaders":4096,"tdp":295},{"name":"Radeon Instinct MI25","vendor":"amd","architecture":"GCN 5.0","generation":"Radeon Instinct(MIx)","memorySize":16.0,"memoryBandwidth":436.2,"memoryType":"HBM2","fp16":24.58,"fp32":12.29,"fp64":0.77,"shaders":4096,"tdp":300},{"name":"Radeon Pro SSG","vendor":"amd","architecture":"GCN 5.0","generation":"Radeon Pro Vega(Vega Series)","memorySize":16.0,"memoryBandwidth":483.8,"memoryType":"HBM2","fp16":24.58,"fp32":12.29,"fp64":0.77,"shaders":4096,"tdp":260},{"name":"Radeon Pro WX 9100","vendor":"amd","architecture":"GCN 5.0","generation":"Radeon Pro Polaris(WX x100)","memorySize":16.0,"memoryBandwidth":483.8,"memoryType":"HBM2","fp16":24.58,"fp32":12.29,"fp64":0.77,"shaders":4096,"tdp":230},{"name":"Radeon RX 6800M","vendor":"amd","architecture":"RDNA 2.0","generation":"Navi Mobile(RX 6000M)","memorySize":12.0,"memoryBandwidth":384.0,"memoryType":"GDDR6","fp16":24.47,"fp32":12.24,"fp64":0.76,"shaders":2560,"tdp":145},{"name":"Radeon PRO W7500","vendor":"amd","architecture":"RDNA 3.0","generation":"Radeon Pro Navi(Navi III Series)","memorySize":8.0,"memoryBandwidth":256.0,"memoryType":"GDDR6","fp16":24.37,"fp32":12.19,"fp64":0.38,"shaders":1792,"tdp":70},{"name":"Xbox Series X 6nm GPU","vendor":"amd","architecture":"RDNA 2.0","generation":"Console GPU(Microsoft)","memorySize":10.0,"memoryBandwidth":560.0,"memoryType":"GDDR6","fp16":24.29,"fp32":12.15,"fp64":0.76,"shaders":3328,"tdp":200},{"name":"Xbox Series X GPU","vendor":"amd","architecture":"RDNA 2.0","generation":"Console GPU(Microsoft)","memorySize":10.0,"memoryBandwidth":560.0,"memoryType":"GDDR6","fp16":24.29,"fp32":12.15,"fp64":0.76,"shaders":3328,"tdp":200},{"name":"Radeon Pro Vega 64X","vendor":"amd","architecture":"GCN 5.0","generation":"Radeon Pro Mac(Vega Series)","memorySize":16.0,"memoryBandwidth":512.0,"memoryType":"HBM2","fp16":24.05,"fp32":12.03,"fp64":0.75,"shaders":4096,"tdp":250},{"name":"Radeon RX 6700","vendor":"amd","architecture":"RDNA 2.0","generation":"Navi II(RX 6000)","memorySize":10.0,"memoryBandwidth":320.0,"memoryType":"GDDR6","fp16":22.58,"fp32":11.29,"fp64":0.71,"shaders":2304,"tdp":175},{"name":"Radeon RX 6750 GRE 10 GB","vendor":"amd","architecture":"RDNA 2.0","generation":"Navi II(RX 6000)","memorySize":10.0,"memoryBandwidth":320.0,"memoryType":"GDDR6","fp16":22.58,"fp32":11.29,"fp64":0.71,"shaders":2304,"tdp":170},{"name":"Radeon Pro Vega 64","vendor":"amd","architecture":"GCN 5.0","generation":"Radeon Pro Mac(Vega Series)","memorySize":16.0,"memoryBandwidth":402.4,"memoryType":"HBM2","fp16":22.12,"fp32":11.06,"fp64":0.69,"shaders":4096,"tdp":250},{"name":"Radeon RX 6700M","vendor":"amd","architecture":"RDNA 2.0","generation":"Navi Mobile(RX 6000M)","memorySize":10.0,"memoryBandwidth":320.0,"memoryType":"GDDR6","fp16":22.12,"fp32":11.06,"fp64":0.69,"shaders":2304,"tdp":135},{"name":"Radeon RX 6650 XT","vendor":"amd","architecture":"RDNA 2.0","generation":"Navi II(RX 6000)","memorySize":8.0,"memoryBandwidth":280.3,"memoryType":"GDDR6","fp16":21.59,"fp32":10.79,"fp64":0.67,"shaders":2048,"tdp":176},{"name":"Radeon Pro V340 16 GB","vendor":"amd","architecture":"GCN 5.0","generation":"Radeon Pro Vega(Vega Series)","memorySize":16.0,"memoryBandwidth":483.8,"memoryType":"HBM2","fp16":21.5,"fp32":10.75,"fp64":0.67,"shaders":3584,"tdp":230},{"name":"Radeon Pro V320","vendor":"amd","architecture":"GCN 5.0","generation":"Radeon Pro Vega(Vega Series)","memorySize":8.0,"memoryBandwidth":483.8,"memoryType":"HBM2","fp16":21.5,"fp32":10.75,"fp64":0.67,"shaders":3584,"tdp":230},{"name":"Radeon Pro V340 8 GB","vendor":"amd","architecture":"GCN 5.0","generation":"Radeon Pro Vega(Vega Series)","memorySize":8.0,"memoryBandwidth":483.8,"memoryType":"HBM2","fp16":21.5,"fp32":10.75,"fp64":0.67,"shaders":3584,"tdp":230},{"name":"Radeon Pro WX 8100","vendor":"amd","architecture":"GCN 5.0","generation":"Radeon Pro Polaris(WX x100)","memorySize":8.0,"memoryBandwidth":512.0,"memoryType":"HBM2","fp16":21.5,"fp32":10.75,"fp64":0.67,"shaders":3584,"tdp":230},{"name":"Radeon Pro WX 8200","vendor":"amd","architecture":"GCN 5.0","generation":"Radeon Pro Polaris(WX x200)","memorySize":8.0,"memoryBandwidth":512.0,"memoryType":"HBM2","fp16":21.5,"fp32":10.75,"fp64":0.67,"shaders":3584,"tdp":230},{"name":"Radeon RX 6600 XT","vendor":"amd","architecture":"RDNA 2.0","generation":"Navi II(RX 6000)","memorySize":8.0,"memoryBandwidth":256.0,"memoryType":"GDDR6","fp16":21.21,"fp32":10.6,"fp64":0.66,"shaders":2048,"tdp":160},{"name":"Radeon RX Vega 56","vendor":"amd","architecture":"GCN 5.0","generation":"Vega(RX Vega)","memorySize":8.0,"memoryBandwidth":409.6,"memoryType":"HBM2","fp16":21.09,"fp32":10.54,"fp64":0.66,"shaders":3584,"tdp":210},{"name":"Radeon Pro W5700X","vendor":"amd","architecture":"RDNA 1.0","generation":"Radeon Pro Mac(Navi Series)","memorySize":16.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":20.89,"fp32":10.44,"fp64":0.65,"shaders":2560,"tdp":205},{"name":"Playstation 5 GPU","vendor":"amd","architecture":"RDNA 2.0","generation":"Console GPU(Sony)","memorySize":16.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":20.58,"fp32":10.29,"fp64":0.64,"shaders":2304,"tdp":180},{"name":"Radeon Pro W6600X","vendor":"amd","architecture":"RDNA 2.0","generation":"Radeon Pro Mac(Navi II Series)","memorySize":8.0,"memoryBandwidth":256.0,"memoryType":"GDDR6","fp16":20.31,"fp32":10.15,"fp64":0.63,"shaders":2048,"tdp":120},{"name":"Radeon RX 5700 XT 50th Anniversary","vendor":"amd","architecture":"RDNA 1.0","generation":"Navi(RX 5000)","memorySize":8.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":20.28,"fp32":10.14,"fp64":0.63,"shaders":2560,"tdp":225},{"name":"Radeon RX 6650M XT","vendor":"amd","architecture":"RDNA 2.0","generation":"Navi Mobile(RX 6000M)","memorySize":8.0,"memoryBandwidth":256.0,"memoryType":"GDDR6","fp16":19.79,"fp32":9.9,"fp64":0.62,"shaders":2048,"tdp":120},{"name":"Radeon RX 5700 XT","vendor":"amd","architecture":"RDNA 1.0","generation":"Navi(RX 5000)","memorySize":8.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":19.51,"fp32":9.75,"fp64":0.61,"shaders":2560,"tdp":225},{"name":"Radeon RX Vega 56 Mobile","vendor":"amd","architecture":"GCN 5.0","generation":"Polaris Mobile(Vega)","memorySize":8.0,"memoryBandwidth":409.6,"memoryType":"HBM2","fp16":18.65,"fp32":9.33,"fp64":0.58,"shaders":3584,"tdp":120},{"name":"Radeon PRO W6600","vendor":"amd","architecture":"RDNA 2.0","generation":"Radeon Pro Navi(Navi II Series)","memorySize":8.0,"memoryBandwidth":224.0,"memoryType":"GDDR6","fp16":18.49,"fp32":9.25,"fp64":0.58,"shaders":1792,"tdp":100},{"name":"Radeon Pro Vega 56","vendor":"amd","architecture":"GCN 5.0","generation":"Radeon Pro Mac(Vega Series)","memorySize":8.0,"memoryBandwidth":402.4,"memoryType":"HBM2","fp16":17.92,"fp32":8.96,"fp64":0.56,"shaders":3584,"tdp":210},{"name":"Radeon RX 6600 LE","vendor":"amd","architecture":"RDNA 2.0","generation":"Navi II(RX 6000)","memorySize":8.0,"memoryBandwidth":224.0,"memoryType":"GDDR6","fp16":17.88,"fp32":8.94,"fp64":0.56,"shaders":1792,"tdp":132},{"name":"Radeon RX 6600","vendor":"amd","architecture":"RDNA 2.0","generation":"Navi II(RX 6000)","memorySize":8.0,"memoryBandwidth":224.0,"memoryType":"GDDR6","fp16":17.86,"fp32":8.93,"fp64":0.56,"shaders":1792,"tdp":132},{"name":"Radeon Pro W5700","vendor":"amd","architecture":"RDNA 1.0","generation":"Radeon Pro Navi(Navi Series)","memorySize":8.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":17.33,"fp32":8.66,"fp64":0.54,"shaders":2304,"tdp":205},{"name":"Radeon RX 6600M","vendor":"amd","architecture":"RDNA 2.0","generation":"Navi Mobile(RX 6000M)","memorySize":8.0,"memoryBandwidth":224.0,"memoryType":"GDDR6","fp16":17.32,"fp32":8.66,"fp64":0.54,"shaders":1792,"tdp":100},{"name":"Radeon RX 6650M","vendor":"amd","architecture":"RDNA 2.0","generation":"Navi Mobile(RX 6000M)","memorySize":8.0,"memoryBandwidth":224.0,"memoryType":"GDDR6","fp16":17.32,"fp32":8.66,"fp64":0.54,"shaders":1792,"tdp":120},{"name":"Radeon RX 6800S","vendor":"amd","architecture":"RDNA 2.0","generation":"Navi Mobile(RX 6000M)","memorySize":8.0,"memoryBandwidth":256.0,"memoryType":"GDDR6","fp16":17.2,"fp32":8.6,"fp64":0.54,"shaders":2048,"tdp":100},{"name":"Ryzen Z1 Extreme GPU","vendor":"amd","architecture":"RDNA 3.0","generation":"Console GPU(AMD)","memorySize":16.0,"memoryBandwidth":51.2,"memoryType":"LPDDR5","fp16":16.59,"fp32":8.29,"fp64":0.52,"shaders":768,"tdp":30},{"name":"Ryzen Z2 GPU","vendor":"amd","architecture":"RDNA 3.0","generation":"Console GPU(AMD)","memorySize":16.0,"memoryBandwidth":119.9,"memoryType":"LPDDR5X","fp16":16.59,"fp32":8.29,"fp64":0.52,"shaders":768,"tdp":28},{"name":"Radeon RX 5700","vendor":"amd","architecture":"RDNA 1.0","generation":"Navi(RX 5000)","memorySize":8.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":15.9,"fp32":7.95,"fp64":0.5,"shaders":2304,"tdp":180},{"name":"Radeon RX 5700M","vendor":"amd","architecture":"RDNA 1.0","generation":"Navi Mobile(RX 5000M)","memorySize":8.0,"memoryBandwidth":384.0,"memoryType":"GDDR6","fp16":15.85,"fp32":7.93,"fp64":0.5,"shaders":2304,"tdp":180},{"name":"Radeon Pro 5700 XT","vendor":"amd","architecture":"RDNA 1.0","generation":"Radeon Pro Mac(Navi Series)","memorySize":16.0,"memoryBandwidth":384.0,"memoryType":"GDDR6","fp16":15.35,"fp32":7.67,"fp64":0.48,"shaders":2560,"tdp":130},{"name":"Radeon Pro V520","vendor":"amd","architecture":"RDNA 1.0","generation":"Radeon Pro Navi(Navi Series)","memorySize":8.0,"memoryBandwidth":512.0,"memoryType":"HBM2","fp16":14.75,"fp32":7.37,"fp64":0.46,"shaders":2304,"tdp":225},{"name":"Radeon Pro Vega 48","vendor":"amd","architecture":"GCN 5.0","generation":"Radeon Pro Mac(Vega Series)","memorySize":8.0,"memoryBandwidth":402.4,"memoryType":"HBM2","fp16":14.75,"fp32":7.37,"fp64":0.46,"shaders":3072},{"name":"Radeon Pro W6600M","vendor":"amd","architecture":"RDNA 2.0","generation":"Radeon Pro Mobile(W6x00M)","memorySize":8.0,"memoryBandwidth":224.0,"memoryType":"GDDR6","fp16":14.58,"fp32":7.29,"fp64":0.46,"shaders":1792,"tdp":90},{"name":"Radeon RX 6700S","vendor":"amd","architecture":"RDNA 2.0","generation":"Navi Mobile(RX 6000M)","memorySize":8.0,"memoryBandwidth":224.0,"memoryType":"GDDR6","fp16":14.34,"fp32":7.17,"fp64":0.45,"shaders":1792,"tdp":80},{"name":"Radeon Pro 5700","vendor":"amd","architecture":"RDNA 1.0","generation":"Radeon Pro Mac(Navi Series)","memorySize":8.0,"memoryBandwidth":384.0,"memoryType":"GDDR6","fp16":12.44,"fp32":6.22,"fp64":0.39,"shaders":2304,"tdp":130},{"name":"Radeon Pro 5600M","vendor":"amd","architecture":"RDNA 1.0","generation":"Radeon Pro Mac(Navi Mobile)","memorySize":8.0,"memoryBandwidth":394.2,"memoryType":"HBM2","fp16":11.71,"fp32":5.86,"fp64":0.37,"shaders":2560,"tdp":50},{"name":"Ryzen AI Z2 Extreme GPU","vendor":"amd","architecture":"RDNA 3.5","generation":"Console GPU(AMD)","memorySize":16.0,"memoryBandwidth":256.0,"memoryType":"LPDDR5X","fp16":11.06,"fp32":5.53,"fp64":0.35,"shaders":1024,"tdp":28},{"name":"Ryzen Z2 Extreme GPU","vendor":"amd","architecture":"RDNA 3.5","generation":"Console GPU(AMD)","memorySize":16.0,"memoryBandwidth":128.0,"memoryType":"LPDDR5X","fp16":11.06,"fp32":5.53,"fp64":0.35,"shaders":1024,"tdp":28},{"name":"Radeon Pro 5500 XT","vendor":"amd","architecture":"RDNA 1.0","generation":"Radeon Pro Mac(Navi Series)","memorySize":8.0,"memoryBandwidth":224.0,"memoryType":"GDDR6","fp16":10.8,"fp32":5.4,"fp64":0.34,"shaders":1536,"tdp":125},{"name":"Radeon Pro W5500X","vendor":"amd","architecture":"RDNA 1.0","generation":"Radeon Pro Mac(Navi Series)","memorySize":8.0,"memoryBandwidth":224.0,"memoryType":"GDDR6","fp16":10.8,"fp32":5.4,"fp64":0.34,"shaders":1536,"tdp":125},{"name":"Radeon Pro W5500","vendor":"amd","architecture":"RDNA 1.0","generation":"Radeon Pro Navi(Navi Series)","memorySize":8.0,"memoryBandwidth":224.0,"memoryType":"GDDR6","fp16":10.45,"fp32":5.22,"fp64":0.33,"shaders":1408,"tdp":125},{"name":"Radeon Pro 5500M","vendor":"amd","architecture":"RDNA 1.0","generation":"Radeon Pro Mac(Navi Mobile)","memorySize":8.0,"memoryBandwidth":192.0,"memoryType":"GDDR6","fp16":8.91,"fp32":4.45,"fp64":0.28,"shaders":1536,"tdp":85},{"name":"Playstation 4 Pro GPU","vendor":"amd","architecture":"GCN 2.0","generation":"Console GPU(Sony)","memorySize":8.0,"memoryBandwidth":217.6,"memoryType":"GDDR5","fp16":8.4,"fp32":4.2,"shaders":2304,"tdp":150},{"name":"Ryzen Z2 Go GPU","vendor":"amd","architecture":"RDNA 2.0","generation":"Console GPU(AMD)","memorySize":16.0,"memoryBandwidth":102.4,"memoryType":"LPDDR5","fp16":8.29,"fp32":4.15,"fp64":0.26,"shaders":768,"tdp":28},{"name":"Xbox Series S GPU","vendor":"amd","architecture":"RDNA 2.0","generation":"Console GPU(Microsoft)","memorySize":8.0,"memoryBandwidth":224.0,"memoryType":"GDDR6","fp16":8.01,"fp32":4.01,"fp64":0.25,"shaders":1280,"tdp":100},{"name":"Zhongshan Subor Z+ GPU","vendor":"amd","architecture":"GCN 5.0","generation":"Console GPU(Zhongshan Subor)","memorySize":8.0,"memoryBandwidth":153.6,"memoryType":"GDDR5","fp16":7.99,"fp32":3.99,"fp64":0.25,"shaders":1536,"tdp":100},{"name":"FirePro S7150","vendor":"amd","architecture":"GCN 3.0","generation":"FirePro Server(Sx100)","memorySize":8.0,"memoryBandwidth":160.0,"memoryType":"GDDR5","fp16":7.54,"fp32":3.77,"fp64":0.24,"shaders":2048,"tdp":150},{"name":"Radeon RX 590","vendor":"amd","architecture":"GCN 4.0","generation":"Polaris(RX 500)","memorySize":8.0,"memoryBandwidth":256.0,"memoryType":"GDDR5","fp16":7.12,"fp32":7.12,"fp64":0.45,"shaders":2304,"tdp":175},{"name":"Radeon RX 590 GME","vendor":"amd","architecture":"GCN 4.0","generation":"Polaris(RX 500)","memorySize":8.0,"memoryBandwidth":256.0,"memoryType":"GDDR5","fp16":6.54,"fp32":6.54,"fp64":0.41,"shaders":2304,"tdp":175},{"name":"Radeon RX 580","vendor":"amd","architecture":"GCN 4.0","generation":"Polaris(RX 500)","memorySize":8.0,"memoryBandwidth":256.0,"memoryType":"GDDR5","fp16":6.17,"fp32":6.17,"fp64":0.39,"shaders":2304,"tdp":185},{"name":"Radeon RX 580X","vendor":"amd","architecture":"GCN 4.0","generation":"Polaris(RX 500X)","memorySize":8.0,"memoryBandwidth":256.0,"memoryType":"GDDR5","fp16":6.17,"fp32":6.17,"fp64":0.39,"shaders":2304,"tdp":185},{"name":"Radeon RX 580G","vendor":"amd","architecture":"GCN 4.0","generation":"Polaris(RX 500)","memorySize":8.0,"memoryBandwidth":256.0,"memoryType":"GDDR5","fp16":6.13,"fp32":6.13,"fp64":0.38,"shaders":2304,"tdp":185},{"name":"Xbox One X GPU","vendor":"amd","architecture":"GCN 2.0","generation":"Console GPU(Microsoft)","memorySize":12.0,"memoryBandwidth":326.4,"memoryType":"GDDR5","fp16":6.0,"fp32":6.0,"shaders":2560,"tdp":150},{"name":"Radeon RX 480","vendor":"amd","architecture":"GCN 4.0","generation":"Arctic Islands(RX 400)","memorySize":8.0,"memoryBandwidth":256.0,"memoryType":"GDDR5","fp16":5.83,"fp32":5.83,"fp64":0.36,"shaders":2304,"tdp":150},{"name":"Radeon RX 580 OEM","vendor":"amd","architecture":"GCN 4.0","generation":"Polaris(RX 500)","memorySize":8.0,"memoryBandwidth":256.0,"memoryType":"GDDR5","fp16":5.83,"fp32":5.83,"fp64":0.36,"shaders":2304,"tdp":150},{"name":"Radeon RX 580X Mobile","vendor":"amd","architecture":"GCN 4.0","generation":"Polaris Mobile(RX M500X)","memorySize":8.0,"memoryBandwidth":256.0,"memoryType":"GDDR5","fp16":5.83,"fp32":5.83,"fp64":0.36,"shaders":2304,"tdp":100},{"name":"Radeon Pro Duo Polaris","vendor":"amd","architecture":"GCN 4.0","generation":"Radeon Pro GCN","memorySize":16.0,"memoryBandwidth":224.0,"memoryType":"GDDR5","fp16":5.73,"fp32":5.73,"fp64":0.36,"shaders":2304,"tdp":250},{"name":"Radeon E9550 MXM","vendor":"amd","architecture":"GCN 4.0","generation":"Embedded(9000)","memorySize":8.0,"memoryBandwidth":160.0,"memoryType":"GDDR5","fp16":5.73,"fp32":5.73,"fp64":0.36,"shaders":2304,"tdp":95},{"name":"Radeon Pro WX 7100","vendor":"amd","architecture":"GCN 4.0","generation":"Radeon Pro Polaris(WX x100)","memorySize":8.0,"memoryBandwidth":224.0,"memoryType":"GDDR5","fp16":5.73,"fp32":5.73,"fp64":0.36,"shaders":2304,"tdp":130},{"name":"Radeon E9560 PCIe","vendor":"amd","architecture":"GCN 4.0","generation":"Embedded(9000)","memorySize":8.0,"memoryBandwidth":224.0,"memoryType":"GDDR5","fp16":5.7,"fp32":5.7,"fp64":0.36,"shaders":2304,"tdp":130},{"name":"Radeon Instinct MI6","vendor":"amd","architecture":"GCN 4.0","generation":"Radeon Instinct(MIx)","memorySize":8.0,"memoryBandwidth":224.0,"memoryType":"GDDR5","fp16":5.68,"fp32":5.68,"fp64":0.36,"shaders":2304,"tdp":150},{"name":"Radeon Pro 580","vendor":"amd","architecture":"GCN 4.0","generation":"Radeon Pro Mac(500 Series)","memorySize":8.0,"memoryBandwidth":217.0,"memoryType":"GDDR5","fp16":5.53,"fp32":5.53,"fp64":0.35,"shaders":2304,"tdp":185},{"name":"Radeon Pro 580X","vendor":"amd","architecture":"GCN 4.0","generation":"Radeon Pro Mac(500X Series)","memorySize":8.0,"memoryBandwidth":218.9,"memoryType":"GDDR5","fp16":5.53,"fp32":5.53,"fp64":0.35,"shaders":2304,"tdp":185},{"name":"Ryzen Z1 GPU","vendor":"amd","architecture":"RDNA 3.0","generation":"Console GPU(AMD)","memorySize":16.0,"memoryBandwidth":51.2,"memoryType":"LPDDR5","fp16":5.12,"fp32":2.56,"fp64":0.16,"shaders":256,"tdp":30},{"name":"Radeon RX 570X","vendor":"amd","architecture":"GCN 4.0","generation":"Polaris(RX 500X)","memorySize":8.0,"memoryBandwidth":224.0,"memoryType":"GDDR5","fp16":5.09,"fp32":5.09,"fp64":0.32,"shaders":2048,"tdp":150},{"name":"Radeon RX 480 Mobile","vendor":"amd","architecture":"GCN 4.0","generation":"Polaris Mobile(RX M400)","memorySize":8.0,"memoryBandwidth":256.0,"memoryType":"GDDR5","fp16":4.96,"fp32":4.96,"fp64":0.31,"shaders":2304,"tdp":100},{"name":"Radeon RX 580 Mobile","vendor":"amd","architecture":"GCN 4.0","generation":"Polaris Mobile(RX M500)","memorySize":8.0,"memoryBandwidth":256.0,"memoryType":"GDDR5","fp16":4.96,"fp32":4.96,"fp64":0.31,"shaders":2304,"tdp":100},{"name":"Radeon RX 570 Mobile","vendor":"amd","architecture":"GCN 4.0","generation":"Polaris Mobile(RX M500)","memorySize":8.0,"memoryBandwidth":211.2,"memoryType":"GDDR5","fp16":4.94,"fp32":4.94,"fp64":0.31,"shaders":2048,"tdp":85},{"name":"Radeon RX 470 Mobile","vendor":"amd","architecture":"GCN 4.0","generation":"Polaris Mobile(RX M400)","memorySize":8.0,"memoryBandwidth":224.0,"memoryType":"GDDR5","fp16":4.4,"fp32":4.4,"fp64":0.27,"shaders":2048,"tdp":85},{"name":"Radeon E8950","vendor":"amd","architecture":"GCN 3.0","generation":"Embedded(8000)","memorySize":8.0,"memoryBandwidth":192.0,"memoryType":"GDDR5","fp16":4.1,"fp32":4.1,"fp64":0.26,"shaders":2048,"tdp":95},{"name":"Radeon E9390 PCIe","vendor":"amd","architecture":"GCN 4.0","generation":"Embedded(9000)","memorySize":8.0,"memoryBandwidth":160.0,"memoryType":"GDDR5","fp16":3.9,"fp32":3.9,"fp64":0.24,"shaders":1792,"tdp":75},{"name":"Radeon Pro WX 5100","vendor":"amd","architecture":"GCN 4.0","generation":"Radeon Pro Polaris(WX x100)","memorySize":8.0,"memoryBandwidth":160.0,"memoryType":"GDDR5","fp16":3.89,"fp32":3.89,"fp64":0.24,"shaders":1792,"tdp":75},{"name":"AeroBox GPU","vendor":"amd","architecture":"GCN 1.0","generation":"Console GPU(Chuwi)","memorySize":8.0,"memoryBandwidth":68.22,"memoryType":"DDR3","fp16":3.53,"fp32":1.76,"fp64":0.11,"shaders":896,"tdp":100},{"name":"FirePro S7150 x2","vendor":"amd","architecture":"GCN 3.0","generation":"FirePro Server(Sx100)","memorySize":8.0,"memoryBandwidth":160.0,"memoryType":"GDDR5","fp16":3.3,"fp32":3.3,"fp64":0.21,"shaders":1792,"tdp":265},{"name":"FirePro W7100","vendor":"amd","architecture":"GCN 3.0","generation":"FirePro GCN(Wx100)","memorySize":8.0,"memoryBandwidth":160.0,"memoryType":"GDDR5","fp16":3.3,"fp32":3.3,"fp64":0.21,"shaders":1792,"tdp":150},{"name":"Ryzen Z2 A GPU","vendor":"amd","architecture":"RDNA 2.0","generation":"Console GPU(AMD)","memorySize":16.0,"memoryBandwidth":102.4,"memoryType":"LPDDR5","fp16":3.28,"fp32":1.64,"fp64":0.1,"shaders":512,"tdp":15},{"name":"Steam Deck GPU","vendor":"amd","architecture":"RDNA 2.0","generation":"Console GPU(Valve)","memorySize":16.0,"memoryBandwidth":176.0,"memoryType":"LPDDR5","fp16":3.28,"fp32":1.64,"fp64":0.1,"shaders":512,"tdp":15},{"name":"Steam Deck OLED GPU","vendor":"amd","architecture":"RDNA 2.0","generation":"Console GPU(Valve)","memorySize":16.0,"memoryBandwidth":176.0,"memoryType":"LPDDR5","fp16":3.28,"fp32":1.64,"fp64":0.1,"shaders":512,"tdp":15},{"name":"FirePro S7100X","vendor":"amd","architecture":"GCN 3.0","generation":"FirePro Mobile(Sx100)","memorySize":8.0,"memoryBandwidth":160.0,"memoryType":"GDDR5","fp16":2.97,"fp32":2.97,"fp64":0.19,"shaders":2048,"tdp":100},{"name":"Radeon R9 M395X","vendor":"amd","architecture":"GCN 3.0","generation":"Gem System(R9 M300)","memorySize":8.0,"memoryBandwidth":160.0,"memoryType":"GDDR5","fp16":2.96,"fp32":2.96,"fp64":0.19,"shaders":2048,"tdp":75},{"name":"Radeon R9 M485X","vendor":"amd","architecture":"GCN 3.0","generation":"Gem System(R9 M400)","memorySize":8.0,"memoryBandwidth":160.0,"memoryType":"GDDR5","fp16":2.96,"fp32":2.96,"fp64":0.19,"shaders":2048,"tdp":250},{"name":"Playstation 4 GPU","vendor":"amd","architecture":"GCN 2.0","generation":"Console GPU(Sony)","memorySize":8.0,"memoryBandwidth":176.0,"memoryType":"GDDR5","fp16":1.84,"fp32":1.84,"shaders":1152,"tdp":75},{"name":"Playstation 4 Slim GPU","vendor":"amd","architecture":"GCN 2.0","generation":"Console GPU(Sony)","memorySize":8.0,"memoryBandwidth":176.0,"memoryType":"GDDR5","fp16":1.84,"fp32":1.84,"shaders":1152,"tdp":75},{"name":"Atari VCS 800 GPU","vendor":"amd","architecture":"GCN 5.0","generation":"Console GPU(Atari)","memorySize":8.0,"memoryBandwidth":38.4,"memoryType":"DDR4","fp16":0.92,"fp32":0.46,"fp64":0.03,"shaders":192,"tdp":15},{"name":"FirePro S9170","vendor":"amd","architecture":"GCN 2.0","generation":"FirePro Server(Sx100)","memorySize":32.0,"memoryBandwidth":320.0,"memoryType":"GDDR5","fp32":5.24,"fp64":2.62,"shaders":2816,"tdp":275},{"name":"FirePro S9150","vendor":"amd","architecture":"GCN 2.0","generation":"FirePro Server(Sx100)","memorySize":16.0,"memoryBandwidth":320.0,"memoryType":"GDDR5","fp32":5.07,"fp64":2.53,"shaders":2816,"tdp":235},{"name":"FirePro W9100","vendor":"amd","architecture":"GCN 2.0","generation":"FirePro GCN(Wx100)","memorySize":16.0,"memoryBandwidth":320.0,"memoryType":"GDDR5","fp32":5.24,"fp64":2.62,"shaders":2816,"tdp":275},{"name":"FirePro S9050","vendor":"amd","architecture":"GCN 1.0","generation":"FirePro Server(Sx000)","memorySize":12.0,"memoryBandwidth":264.0,"memoryType":"GDDR5","fp32":3.23,"fp64":0.81,"shaders":1792,"tdp":225},{"name":"FirePro S9100","vendor":"amd","architecture":"GCN 2.0","generation":"FirePro Server(Sx100)","memorySize":12.0,"memoryBandwidth":320.0,"memoryType":"GDDR5","fp32":4.22,"fp64":2.11,"shaders":2560,"tdp":225},{"name":"FirePro W8100","vendor":"amd","architecture":"GCN 2.0","generation":"FirePro GCN(Wx100)","memorySize":8.0,"memoryBandwidth":320.0,"memoryType":"GDDR5","fp32":4.22,"fp64":2.11,"shaders":2560,"tdp":220},{"name":"Radeon Pro WX 7100 Mobile","vendor":"amd","architecture":"GCN 4.0","generation":"Radeon Pro Mobile(WX x100)","memorySize":8.0,"memoryBandwidth":160.0,"memoryType":"GDDR5","fp32":5.73,"fp64":0.36,"shaders":2304,"tdp":130},{"name":"Radeon Pro WX 7130 Mobile","vendor":"amd","architecture":"GCN 4.0","generation":"Radeon Pro Mobile(WX x100)","memorySize":8.0,"memoryBandwidth":160.0,"memoryType":"GDDR5","fp32":5.73,"fp64":0.36,"shaders":2304,"tdp":130},{"name":"Radeon R9 390","vendor":"amd","architecture":"GCN 2.0","generation":"Pirate Islands(R9 300)","memorySize":8.0,"memoryBandwidth":384.0,"memoryType":"GDDR5","fp32":5.12,"fp64":0.64,"shaders":2560,"tdp":275},{"name":"Radeon R9 390 X2","vendor":"amd","architecture":"GCN 2.0","generation":"Pirate Islands(R9 300)","memorySize":8.0,"memoryBandwidth":345.6,"memoryType":"GDDR5","fp32":5.12,"fp64":0.64,"shaders":2560,"tdp":580},{"name":"Radeon R9 390X","vendor":"amd","architecture":"GCN 2.0","generation":"Pirate Islands(R9 300)","memorySize":8.0,"memoryBandwidth":384.0,"memoryType":"GDDR5","fp32":5.91,"fp64":0.74,"shaders":2816,"tdp":275},{"name":"Xbox One S GPU","vendor":"amd","architecture":"GCN 1.0","generation":"Console GPU(Microsoft)","memorySize":8.0,"memoryBandwidth":68.22,"memoryType":"DDR3","fp32":1.4,"shaders":768,"tdp":95}] diff --git a/internal/gpudb/data/gcp.json b/internal/gpudb/data/gcp.json new file mode 100644 index 0000000..82fa1ba --- /dev/null +++ b/internal/gpudb/data/gcp.json @@ -0,0 +1,18 @@ +[ + {"id":"nvidia-l4","name":"L4","hfName":"L4","series":"G2","memoryGB":24,"machineHint":"g2-standard-4","attach":"machine","spotHourlyApprox":0.35}, + {"id":"nvidia-tesla-t4","name":"T4","hfName":"Tesla T4","series":"N1","memoryGB":16,"machineHint":"n1-standard-4","attach":"accelerator","spotHourlyApprox":0.14}, + {"id":"nvidia-tesla-t4-vws","name":"T4 VWS","hfName":"Tesla T4","series":"N1","memoryGB":16,"machineHint":"n1-standard-4","attach":"accelerator","spotHourlyApprox":0.14}, + {"id":"nvidia-tesla-v100","name":"V100","hfName":"Tesla V100","series":"N1","memoryGB":16,"machineHint":"n1-standard-8","attach":"accelerator","spotHourlyApprox":0.74}, + {"id":"nvidia-tesla-p4","name":"P4","hfName":"Tesla P4","series":"N1","memoryGB":8,"machineHint":"n1-standard-4","attach":"accelerator","spotHourlyApprox":0.27}, + {"id":"nvidia-tesla-p100","name":"P100","hfName":"Tesla P100","series":"N1","memoryGB":16,"machineHint":"n1-standard-8","attach":"accelerator","spotHourlyApprox":0.43}, + {"id":"nvidia-tesla-a100","name":"A100 40GB","hfName":"A100 SXM4 40 GB","series":"A2","memoryGB":40,"machineHint":"a2-highgpu-1g","attach":"machine","spotHourlyApprox":1.57}, + {"id":"nvidia-a100-80gb","name":"A100 80GB","hfName":"A100 SXM4 80 GB","series":"A2","memoryGB":80,"machineHint":"a2-ultragpu-1g","attach":"machine","spotHourlyApprox":2.90}, + {"id":"nvidia-h100-80gb","name":"H100 80GB","hfName":"H100 SXM5 80 GB","series":"A3","memoryGB":80,"machineHint":"a3-highgpu-1g","attach":"machine","spotHourlyApprox":5.50}, + {"id":"nvidia-h100-mega-80gb","name":"H100 Mega 80GB","hfName":"H100 SXM5 80 GB","series":"A3","memoryGB":80,"machineHint":"a3-megagpu-8g","attach":"machine","spotHourlyApprox":5.50}, + {"id":"nvidia-h200-141gb","name":"H200 141GB","hfName":"H200 SXM 141 GB","series":"A3","memoryGB":141,"machineHint":"a3-ultragpu-8g","attach":"machine","spotHourlyApprox":8.00}, + {"id":"nvidia-b200","name":"B200","hfName":"B200","series":"A4","memoryGB":180,"machineHint":"a4-highgpu-8g","attach":"machine","spotHourlyApprox":12.00}, + {"id":"nvidia-gb200","name":"GB200","hfName":"B200","series":"A4X","memoryGB":192,"machineHint":"a4x","attach":"machine","spotHourlyApprox":0}, + {"id":"nvidia-gb300","name":"GB300","hfName":"B300","series":"A4X","memoryGB":288,"machineHint":"a4x-max","attach":"machine","spotHourlyApprox":0}, + {"id":"nvidia-rtx-pro-6000","name":"RTX PRO 6000","hfName":"RTX PRO 6000 Blackwell Server Edition","series":"G4","memoryGB":96,"machineHint":"g4-standard-48","attach":"machine","spotHourlyApprox":2.50}, + {"id":"nvidia-l4-vws","name":"L4 VWS","hfName":"L4","series":"G2","memoryGB":24,"machineHint":"g2-standard-4","attach":"machine","spotHourlyApprox":0.35} +] diff --git a/internal/gpudb/data/nvidia.json b/internal/gpudb/data/nvidia.json new file mode 100644 index 0000000..aa73475 --- /dev/null +++ b/internal/gpudb/data/nvidia.json @@ -0,0 +1 @@ +[{"name":"B300","architecture":"Blackwell Ultra","generation":"Server Blackwell(Bxx)","memorySize":144.0,"memoryBandwidth":4099.999999999999,"memoryType":"HBM3e","fp16":1231.8,"fp32":76.99,"fp64":1.2,"tensorCores":592,"cuda":"10.3","shaders":18944,"tdp":1400},{"name":"B200","architecture":"Blackwell","generation":"Server Blackwell(Bxx)","memorySize":90.0,"memoryBandwidth":4099.999999999999,"memoryType":"HBM3e","fp16":1191.2,"fp32":74.45,"fp64":37.22,"tensorCores":592,"cuda":"10.0","shaders":18944,"tdp":1000},{"name":"H200 SXM 141 GB","architecture":"Hopper","generation":"Server Hopper(Hxx)","memorySize":141.0,"memoryBandwidth":4890.0,"memoryType":"HBM3e","fp16":267.6,"fp32":66.91,"fp64":33.45,"tensorCores":528,"cuda":"9.0","shaders":16896,"tdp":700},{"name":"H100 SXM5 96 GB","architecture":"Hopper","generation":"Server Hopper(Hxx)","memorySize":96.0,"memoryBandwidth":3360.0,"memoryType":"HBM3","fp16":267.6,"fp32":66.91,"fp64":33.45,"tensorCores":528,"cuda":"9.0","shaders":16896,"tdp":700},{"name":"H100 SXM5 94 GB","architecture":"Hopper","generation":"Server Hopper(Hxx)","memorySize":94.0,"memoryBandwidth":3360.0,"memoryType":"HBM3","fp16":267.6,"fp32":66.91,"fp64":33.45,"tensorCores":528,"cuda":"9.0","shaders":16896,"tdp":700},{"name":"H100 SXM5 80 GB","architecture":"Hopper","generation":"Server Hopper(Hxx)","memorySize":80.0,"memoryBandwidth":3360.0,"memoryType":"HBM3","fp16":267.6,"fp32":66.91,"fp64":33.45,"tensorCores":528,"cuda":"9.0","shaders":16896,"tdp":700},{"name":"H100 SXM5 64 GB","architecture":"Hopper","generation":"Server Hopper(Hxx)","memorySize":64.0,"memoryBandwidth":2020.0,"memoryType":"HBM3","fp16":267.6,"fp32":66.91,"fp64":33.45,"tensorCores":528,"cuda":"9.0","shaders":16896,"tdp":700},{"name":"H100 PCIe 96 GB","architecture":"Hopper","generation":"Server Hopper(Hxx)","memorySize":96.0,"memoryBandwidth":3360.0,"memoryType":"HBM3","fp16":248.3,"fp32":62.08,"fp64":31.04,"tensorCores":528,"cuda":"9.0","shaders":16896,"tdp":700},{"name":"H200 NVL","architecture":"Hopper","generation":"Server Hopper(Hxx)","memorySize":141.0,"memoryBandwidth":4890.0,"memoryType":"HBM3e","fp16":241.3,"fp32":60.32,"fp64":30.16,"tensorCores":528,"cuda":"9.0","shaders":16896,"tdp":600},{"name":"H100 NVL 94 GB","architecture":"Hopper","generation":"Server Hopper(Hxx)","memorySize":94.0,"memoryBandwidth":3940.0,"memoryType":"HBM3","fp16":241.3,"fp32":60.32,"fp64":30.16,"tensorCores":528,"cuda":"9.0","shaders":16896,"tdp":400},{"name":"H800 SXM5","architecture":"Hopper","generation":"Server Hopper(Hxx)","memorySize":80.0,"memoryBandwidth":3360.0,"memoryType":"HBM3","fp16":237.2,"fp32":59.3,"fp64":29.65,"tensorCores":528,"cuda":"9.0","shaders":16896,"tdp":700},{"name":"H100 CNX","architecture":"Hopper","generation":"Server Hopper(Hxx)","memorySize":80.0,"memoryBandwidth":2040.0,"memoryType":"HBM2e","fp16":215.4,"fp32":53.84,"fp64":26.92,"tensorCores":456,"cuda":"9.0","shaders":14592,"tdp":350},{"name":"H100 PCIe 80 GB","architecture":"Hopper","generation":"Server Hopper(Hxx)","memorySize":80.0,"memoryBandwidth":2040.0,"memoryType":"HBM2e","fp16":204.9,"fp32":51.22,"fp64":25.61,"tensorCores":456,"cuda":"9.0","shaders":14592,"tdp":350},{"name":"H800 PCIe 80 GB","architecture":"Hopper","generation":"Server Hopper(Hxx)","memorySize":80.0,"memoryBandwidth":2040.0,"memoryType":"HBM2e","fp16":204.9,"fp32":51.22,"fp64":25.61,"tensorCores":456,"cuda":"9.0","shaders":14592,"tdp":350},{"name":"RTX PRO 6000 Blackwell","architecture":"Blackwell 2.0","generation":"Blackwell PRO W(x000)","memorySize":96.0,"memoryBandwidth":1790.0,"memoryType":"GDDR7","fp16":126.0,"fp32":126.0,"fp64":1.97,"tensorCores":752,"cuda":"12.0","shaders":24064,"tdp":600},{"name":"RTX PRO 6000 Blackwell Server","architecture":"Blackwell 2.0","generation":"Server Blackwell(Bxx)","memorySize":96.0,"memoryBandwidth":1790.0,"memoryType":"GDDR7","fp16":126.0,"fp32":126.0,"fp64":1.97,"tensorCores":752,"cuda":"12.0","shaders":24064,"tdp":600},{"name":"RTX PRO 6000D Blackwell Max-Q","architecture":"Blackwell 2.0","generation":"Blackwell PRO W(x000)","memorySize":96.0,"memoryBandwidth":1790.0,"memoryType":"GDDR7","fp16":110.1,"fp32":110.1,"fp64":1.72,"tensorCores":752,"cuda":"12.0","shaders":24064,"tdp":300},{"name":"RTX PRO 6000 Blackwell Max-Q","architecture":"Blackwell 2.0","generation":"Blackwell PRO W(x000)","memorySize":96.0,"memoryBandwidth":1790.0,"memoryType":"GDDR7","fp16":109.7,"fp32":109.7,"fp64":1.72,"tensorCores":752,"cuda":"12.0","shaders":24064,"tdp":300},{"name":"GeForce RTX 5090","architecture":"Blackwell 2.0","generation":"GeForce 50","memorySize":32.0,"memoryBandwidth":1790.0,"memoryType":"GDDR7","fp16":104.8,"fp32":104.8,"fp64":1.64,"tensorCores":680,"cuda":"12.0","shaders":21760,"tdp":575},{"name":"GeForce RTX 5090 D","architecture":"Blackwell 2.0","generation":"GeForce 50","memorySize":32.0,"memoryBandwidth":1790.0,"memoryType":"GDDR7","fp16":104.8,"fp32":104.8,"fp64":1.64,"tensorCores":680,"cuda":"12.0","shaders":21760,"tdp":575},{"name":"GeForce RTX 5090 D V2","architecture":"Blackwell 2.0","generation":"GeForce 50","memorySize":24.0,"memoryBandwidth":1340.0,"memoryType":"GDDR7","fp16":104.8,"fp32":104.8,"fp64":1.64,"tensorCores":680,"cuda":"12.0","shaders":21760,"tdp":575},{"name":"RTX 6000D","architecture":"Blackwell 2.0","generation":"Blackwell PRO W(x000)","memorySize":84.0,"memoryBandwidth":1570.0,"memoryType":"GDDR7","fp16":97.04,"fp32":97.04,"fp64":1.52,"tensorCores":624,"cuda":"12.0","shaders":19968,"tdp":600},{"name":"L40S","architecture":"Ada Lovelace","generation":"Server Ada(Lxx)","memorySize":48.0,"memoryBandwidth":864.0,"memoryType":"GDDR6","fp16":91.61,"fp32":91.61,"fp64":1.43,"tensorCores":568,"cuda":"8.9","shaders":18176,"tdp":300},{"name":"RTX 6000 Ada Generation","architecture":"Ada Lovelace","generation":"Workstation Ada(x000A)","memorySize":48.0,"memoryBandwidth":960.0,"memoryType":"GDDR6","fp16":91.06,"fp32":91.06,"fp64":1.42,"tensorCores":568,"cuda":"8.9","shaders":18176,"tdp":300},{"name":"L40","architecture":"Ada Lovelace","generation":"Server Ada(Lxx)","memorySize":48.0,"memoryBandwidth":864.0,"memoryType":"GDDR6","fp16":90.52,"fp32":90.52,"fp64":1.41,"tensorCores":568,"cuda":"8.9","shaders":18176,"tdp":300},{"name":"L40 CNX","architecture":"Ada Lovelace","generation":"Server Ada(Lxx)","memorySize":24.0,"memoryBandwidth":864.0,"memoryType":"GDDR6","fp16":89.97,"fp32":89.97,"fp64":1.41,"tensorCores":568,"cuda":"8.9","shaders":18176,"tdp":300},{"name":"L40G","architecture":"Ada Lovelace","generation":"Server Ada(Lxx)","memorySize":24.0,"memoryBandwidth":864.0,"memoryType":"GDDR6","fp16":89.97,"fp32":89.97,"fp64":1.41,"tensorCores":568,"cuda":"8.9","shaders":18176,"tdp":300},{"name":"GeForce RTX 4090","architecture":"Ada Lovelace","generation":"GeForce 40","memorySize":24.0,"memoryBandwidth":1010.0,"memoryType":"GDDR6X","fp16":82.58,"fp32":82.58,"fp64":1.29,"tensorCores":512,"cuda":"8.9","shaders":16384,"tdp":450},{"name":"A100X","architecture":"Ampere","generation":"Server Ampere(Axx)","memorySize":80.0,"memoryBandwidth":2040.0,"memoryType":"HBM2e","fp16":79.63,"fp32":19.91,"fp64":9.95,"tensorCores":432,"cuda":"8.0","shaders":6912,"tdp":300},{"name":"A100 PCIe 80 GB","architecture":"Ampere","generation":"Server Ampere(Axx)","memorySize":80.0,"memoryBandwidth":1940.0,"memoryType":"HBM2e","fp16":77.97,"fp32":19.49,"fp64":9.75,"tensorCores":432,"cuda":"8.0","shaders":6912,"tdp":300},{"name":"A100 SXM4 80 GB","architecture":"Ampere","generation":"Server Ampere(Axx)","memorySize":80.0,"memoryBandwidth":2040.0,"memoryType":"HBM2e","fp16":77.97,"fp32":19.49,"fp64":9.75,"tensorCores":432,"cuda":"8.0","shaders":6912,"tdp":400},{"name":"A800 PCIe 80 GB","architecture":"Ampere","generation":"Server Ampere(Axx)","memorySize":80.0,"memoryBandwidth":1940.0,"memoryType":"HBM2e","fp16":77.97,"fp32":19.49,"fp64":9.75,"tensorCores":432,"cuda":"8.0","shaders":6912,"tdp":250},{"name":"A800 SXM4 80 GB","architecture":"Ampere","generation":"Server Ampere(Axx)","memorySize":80.0,"memoryBandwidth":2040.0,"memoryType":"HBM2e","fp16":77.97,"fp32":19.49,"fp64":9.75,"tensorCores":432,"cuda":"8.0","shaders":6912,"tdp":400},{"name":"A100 PCIe 40 GB","architecture":"Ampere","generation":"Server Ampere(Axx)","memorySize":40.0,"memoryBandwidth":1560.0,"memoryType":"HBM2e","fp16":77.97,"fp32":19.49,"fp64":9.75,"tensorCores":432,"cuda":"8.0","shaders":6912,"tdp":250},{"name":"A100 SXM4 40 GB","architecture":"Ampere","generation":"Server Ampere(Axx)","memorySize":40.0,"memoryBandwidth":1560.0,"memoryType":"HBM2e","fp16":77.97,"fp32":19.49,"fp64":9.75,"tensorCores":432,"cuda":"8.0","shaders":6912,"tdp":400},{"name":"A800 PCIe 40 GB","architecture":"Ampere","generation":"Server Ampere(Axx)","memorySize":40.0,"memoryBandwidth":1560.0,"memoryType":"HBM2e","fp16":77.97,"fp32":19.49,"fp64":9.75,"tensorCores":432,"cuda":"8.0","shaders":6912,"tdp":250},{"name":"GRID A100A","architecture":"Ampere","generation":"GRID(Ax)","memorySize":32.0,"memoryBandwidth":1870.0,"memoryType":"HBM2e","fp16":77.97,"fp32":19.49,"fp64":9.75,"tensorCores":432,"cuda":"8.0","shaders":6912,"tdp":400},{"name":"GeForce RTX 4090 D","architecture":"Ada Lovelace","generation":"GeForce 40","memorySize":24.0,"memoryBandwidth":1010.0,"memoryType":"GDDR6X","fp16":73.54,"fp32":73.54,"fp64":1.15,"tensorCores":456,"cuda":"8.9","shaders":14592,"tdp":425},{"name":"DRIVE A100 PROD","architecture":"Ampere","generation":"DRIVE(Axx)","memorySize":32.0,"memoryBandwidth":1870.0,"memoryType":"HBM2e","fp16":69.67,"fp32":17.42,"fp64":8.71,"tensorCores":432,"cuda":"8.0","shaders":6912,"tdp":400},{"name":"RTX 5880 Ada Generation","architecture":"Ada Lovelace","generation":"Workstation Ada(x000A)","memorySize":48.0,"memoryBandwidth":864.0,"memoryType":"GDDR6","fp16":69.27,"fp32":69.27,"fp64":1.08,"tensorCores":440,"cuda":"8.9","shaders":14080,"tdp":285},{"name":"RTX PRO 5000 72 GB Blackwell","architecture":"Blackwell 2.0","generation":"Blackwell PRO W(x000)","memorySize":72.0,"memoryBandwidth":1340.0,"memoryType":"GDDR7","fp16":66.94,"fp32":66.94,"fp64":1.05,"tensorCores":440,"cuda":"12.0","shaders":14080,"tdp":300},{"name":"RTX PRO 5000 Blackwell","architecture":"Blackwell 2.0","generation":"Blackwell PRO W(x000)","memorySize":48.0,"memoryBandwidth":1340.0,"memoryType":"GDDR7","fp16":66.94,"fp32":66.94,"fp64":1.05,"tensorCores":440,"cuda":"12.0","shaders":14080,"tdp":300},{"name":"RTX 5000 Ada Generation","architecture":"Ada Lovelace","generation":"Workstation Ada(x000A)","memorySize":32.0,"memoryBandwidth":576.0,"memoryType":"GDDR6","fp16":65.28,"fp32":65.28,"fp64":1.02,"tensorCores":400,"cuda":"8.9","shaders":12800,"tdp":250},{"name":"Tesla T4","architecture":"Turing","generation":"Tesla Turing(Txx)","memorySize":16.0,"memoryBandwidth":320.0,"memoryType":"GDDR6","fp16":65.13,"fp32":8.14,"fp64":0.25,"tensorCores":320,"cuda":"7.5","shaders":2560,"tdp":70},{"name":"Tesla T4G","architecture":"Turing","generation":"Tesla Turing(Txx)","memorySize":16.0,"memoryBandwidth":320.0,"memoryType":"GDDR6","fp16":65.13,"fp32":8.14,"fp64":0.25,"tensorCores":320,"cuda":"7.5","shaders":2560,"tdp":70},{"name":"L20","architecture":"Ada Lovelace","generation":"Server Ada(Lxx)","memorySize":48.0,"memoryBandwidth":864.0,"memoryType":"GDDR6","fp16":59.35,"fp32":59.35,"fp64":0.93,"tensorCores":368,"cuda":"8.9","shaders":11776,"tdp":275},{"name":"GeForce RTX 5080","architecture":"Blackwell 2.0","generation":"GeForce 50","memorySize":16.0,"memoryBandwidth":960.0,"memoryType":"GDDR7","fp16":56.28,"fp32":56.28,"fp64":0.88,"tensorCores":336,"cuda":"12.0","shaders":10752,"tdp":360},{"name":"GRID A100B","architecture":"Ampere","generation":"GRID(Ax)","memorySize":48.0,"memoryBandwidth":1870.0,"memoryType":"HBM2e","fp16":55.57,"fp32":13.89,"fp64":6.95,"tensorCores":432,"cuda":"8.0","shaders":6912,"tdp":400},{"name":"GeForce RTX 4080 SUPER","architecture":"Ada Lovelace","generation":"GeForce 40","memorySize":16.0,"memoryBandwidth":736.3,"memoryType":"GDDR6X","fp16":52.22,"fp32":52.22,"fp64":0.82,"tensorCores":320,"cuda":"8.9","shaders":10240,"tdp":320},{"name":"Jetson T5000","architecture":"Blackwell","generation":"Server Blackwell(Bxx)","memorySize":128.0,"memoryBandwidth":273.2,"memoryType":"LPDDR5X","fp16":51.71,"fp32":12.93,"fp64":6.46,"tensorCores":96,"cuda":"11.0","shaders":2560,"tdp":40},{"name":"RTX PRO 4500 Blackwell","architecture":"Blackwell 2.0","generation":"Blackwell PRO W(x000)","memorySize":32.0,"memoryBandwidth":896.0,"memoryType":"GDDR7","fp16":50.53,"fp32":50.53,"fp64":0.79,"tensorCores":328,"cuda":"12.0","shaders":10496,"tdp":200},{"name":"CMP 170HX 10 GB","architecture":"Ampere","generation":"Mining GPUs","memorySize":10.0,"memoryBandwidth":1560.0,"memoryType":"HBM2e","fp16":50.53,"fp32":12.63,"fp64":6.32,"tensorCores":280,"cuda":"8.0","shaders":4480,"tdp":250},{"name":"CMP 170HX 8 GB","architecture":"Ampere","generation":"Mining GPUs","memorySize":8.0,"memoryBandwidth":1490.0,"memoryType":"HBM2e","fp16":50.53,"fp32":12.63,"fp64":6.32,"tensorCores":280,"cuda":"8.0","shaders":4480,"tdp":250},{"name":"GeForce RTX 4080","architecture":"Ada Lovelace","generation":"GeForce 40","memorySize":16.0,"memoryBandwidth":716.8,"memoryType":"GDDR6X","fp16":48.74,"fp32":48.74,"fp64":0.76,"tensorCores":304,"cuda":"8.9","shaders":9728,"tdp":320},{"name":"GeForce RTX 4070 Ti SUPER","architecture":"Ada Lovelace","generation":"GeForce 40","memorySize":16.0,"memoryBandwidth":672.3,"memoryType":"GDDR6X","fp16":44.1,"fp32":44.1,"fp64":0.69,"tensorCores":264,"cuda":"8.9","shaders":8448,"tdp":285},{"name":"GeForce RTX 4070 Ti SUPER AD102","architecture":"Ada Lovelace","generation":"GeForce 40","memorySize":16.0,"memoryBandwidth":672.3,"memoryType":"GDDR6X","fp16":44.1,"fp32":44.1,"fp64":0.69,"tensorCores":264,"cuda":"8.9","shaders":8448,"tdp":285},{"name":"GeForce RTX 5070 Ti","architecture":"Blackwell 2.0","generation":"GeForce 50","memorySize":16.0,"memoryBandwidth":896.0,"memoryType":"GDDR7","fp16":43.94,"fp32":43.94,"fp64":0.69,"tensorCores":280,"cuda":"12.0","shaders":8960,"tdp":300},{"name":"RTX 5000 Mobile Ada Generation","architecture":"Ada Lovelace","generation":"Ada-MW(x000A)","memorySize":16.0,"memoryBandwidth":576.0,"memoryType":"GDDR6","fp16":41.15,"fp32":41.15,"fp64":0.64,"tensorCores":304,"cuda":"8.9","shaders":9728,"tdp":120},{"name":"GeForce RTX 4070 Ti","architecture":"Ada Lovelace","generation":"GeForce 40","memorySize":12.0,"memoryBandwidth":504.2,"memoryType":"GDDR6X","fp16":40.09,"fp32":40.09,"fp64":0.63,"tensorCores":240,"cuda":"8.9","shaders":7680,"tdp":285},{"name":"GeForce RTX 3090 Ti","architecture":"Ampere","generation":"GeForce 30","memorySize":24.0,"memoryBandwidth":1010.0,"memoryType":"GDDR6X","fp16":40.0,"fp32":40.0,"fp64":0.62,"tensorCores":336,"cuda":"8.6","shaders":10752,"tdp":450},{"name":"RTX 4500 Ada Generation","architecture":"Ada Lovelace","generation":"Workstation Ada(x000A)","memorySize":24.0,"memoryBandwidth":432.0,"memoryType":"GDDR6","fp16":39.63,"fp32":39.63,"fp64":0.62,"tensorCores":240,"cuda":"8.9","shaders":7680,"tdp":210},{"name":"RTX A6000","architecture":"Ampere","generation":"Workstation Ampere(Ax000)","memorySize":48.0,"memoryBandwidth":768.0,"memoryType":"GDDR6","fp16":38.71,"fp32":38.71,"fp64":0.6,"tensorCores":336,"cuda":"8.6","shaders":10752,"tdp":300},{"name":"A40 PCIe","architecture":"Ampere","generation":"Server Ampere(Axx)","memorySize":48.0,"memoryBandwidth":695.8,"memoryType":"GDDR6","fp16":37.42,"fp32":37.42,"fp64":0.58,"tensorCores":336,"cuda":"8.6","shaders":10752,"tdp":300},{"name":"RTX PRO 4000 Blackwell","architecture":"Blackwell 2.0","generation":"Blackwell PRO W(x000)","memorySize":24.0,"memoryBandwidth":672.0,"memoryType":"GDDR7","fp16":36.83,"fp32":36.83,"fp64":0.58,"tensorCores":280,"cuda":"12.0","shaders":8960,"tdp":140},{"name":"GeForce RTX 3090","architecture":"Ampere","generation":"GeForce 30","memorySize":24.0,"memoryBandwidth":936.2,"memoryType":"GDDR6X","fp16":35.58,"fp32":35.58,"fp64":0.56,"tensorCores":328,"cuda":"8.6","shaders":10496,"tdp":350},{"name":"GeForce RTX 4070 SUPER","architecture":"Ada Lovelace","generation":"GeForce 40","memorySize":12.0,"memoryBandwidth":504.2,"memoryType":"GDDR6X","fp16":35.48,"fp32":35.48,"fp64":0.55,"tensorCores":224,"cuda":"8.9","shaders":7168,"tdp":220},{"name":"RTX A5500","architecture":"Ampere","generation":"Workstation Ampere(Ax000)","memorySize":24.0,"memoryBandwidth":768.0,"memoryType":"GDDR6","fp16":34.1,"fp32":34.1,"fp64":0.53,"tensorCores":320,"cuda":"8.6","shaders":10240,"tdp":230},{"name":"GeForce RTX 3080 Ti 20 GB","architecture":"Ampere","generation":"GeForce 30","memorySize":20.0,"memoryBandwidth":760.3,"memoryType":"GDDR6X","fp16":34.1,"fp32":34.1,"fp64":0.53,"tensorCores":320,"cuda":"8.6","shaders":10240,"tdp":350},{"name":"GeForce RTX 3080 Ti","architecture":"Ampere","generation":"GeForce 30","memorySize":12.0,"memoryBandwidth":912.4,"memoryType":"GDDR6X","fp16":34.1,"fp32":34.1,"fp64":0.53,"tensorCores":320,"cuda":"8.6","shaders":10240,"tdp":350},{"name":"GeForce RTX 4090 Mobile","architecture":"Ada Lovelace","generation":"GeForce 40 Mobile","memorySize":16.0,"memoryBandwidth":576.0,"memoryType":"GDDR6","fp16":32.98,"fp32":32.98,"fp64":0.52,"tensorCores":304,"cuda":"8.9","shaders":9728,"tdp":120},{"name":"RTX 5000 Embedded Ada Generation","architecture":"Ada Lovelace","generation":"Ada-MW(x000A)","memorySize":16.0,"memoryBandwidth":576.0,"memoryType":"GDDR6","fp16":32.69,"fp32":32.69,"fp64":0.51,"tensorCores":304,"cuda":"8.9","shaders":9728,"tdp":120},{"name":"RTX 5000 Embedded Ada Generation X2","architecture":"Ada Lovelace","generation":"Ada-MW(x000A)","memorySize":16.0,"memoryBandwidth":576.0,"memoryType":"GDDR6","fp16":32.69,"fp32":32.69,"fp64":0.51,"tensorCores":304,"cuda":"8.9","shaders":9728,"tdp":150},{"name":"RTX 5000 Max-Q Ada Generation","architecture":"Ada Lovelace","generation":"Ada-MW(x000A)","memorySize":16.0,"memoryBandwidth":576.0,"memoryType":"GDDR6","fp16":32.69,"fp32":32.69,"fp64":0.51,"tensorCores":304,"cuda":"8.9","shaders":9728,"tdp":120},{"name":"Quadro RTX 8000","architecture":"Turing","generation":"Quadro Turing(Tx000)","memorySize":48.0,"memoryBandwidth":672.0,"memoryType":"GDDR6","fp16":32.62,"fp32":16.31,"fp64":0.51,"tensorCores":576,"cuda":"7.5","shaders":4608,"tdp":260},{"name":"Quadro RTX 6000","architecture":"Turing","generation":"Quadro Turing(Tx000)","memorySize":24.0,"memoryBandwidth":672.0,"memoryType":"GDDR6","fp16":32.62,"fp32":16.31,"fp64":0.51,"tensorCores":576,"cuda":"7.5","shaders":4608,"tdp":260},{"name":"TITAN RTX","architecture":"Turing","generation":"GeForce 20","memorySize":24.0,"memoryBandwidth":672.0,"memoryType":"GDDR6","fp16":32.62,"fp32":16.31,"fp64":0.51,"tensorCores":576,"cuda":"7.5","shaders":4608,"tdp":280},{"name":"GeForce RTX 5090 Mobile","architecture":"Blackwell 2.0","generation":"GeForce 50 Mobile","memorySize":24.0,"memoryBandwidth":896.0,"memoryType":"GDDR7","fp16":31.8,"fp32":31.8,"fp64":0.5,"tensorCores":328,"cuda":"12.0","shaders":10496,"tdp":95},{"name":"A10G","architecture":"Ampere","generation":"Server Ampere(Axx)","memorySize":24.0,"memoryBandwidth":600.2,"memoryType":"GDDR6","fp16":31.52,"fp32":31.52,"fp64":0.98,"tensorCores":288,"cuda":"8.6","shaders":9216,"tdp":150},{"name":"A10 PCIe","architecture":"Ampere","generation":"Server Ampere(Axx)","memorySize":24.0,"memoryBandwidth":600.2,"memoryType":"GDDR6","fp16":31.24,"fp32":31.24,"fp64":0.98,"tensorCores":288,"cuda":"8.6","shaders":9216,"tdp":150},{"name":"Jetson T4000","architecture":"Blackwell","generation":"Server Blackwell(Bxx)","memorySize":64.0,"memoryBandwidth":273.2,"memoryType":"LPDDR5X","fp16":31.03,"fp32":7.76,"fp64":3.88,"tensorCores":64,"cuda":"11.0","shaders":1536,"tdp":40},{"name":"GeForce RTX 5070","architecture":"Blackwell 2.0","generation":"GeForce 50","memorySize":12.0,"memoryBandwidth":672.0,"memoryType":"GDDR7","fp16":30.87,"fp32":30.87,"fp64":0.48,"tensorCores":192,"cuda":"12.0","shaders":6144,"tdp":250},{"name":"GeForce RTX 3080 12 GB","architecture":"Ampere","generation":"GeForce 30","memorySize":12.0,"memoryBandwidth":912.4,"memoryType":"GDDR6X","fp16":30.64,"fp32":30.64,"fp64":0.48,"tensorCores":280,"cuda":"8.6","shaders":8960,"tdp":350},{"name":"L4","architecture":"Ada Lovelace","generation":"Server Ada(Lxx)","memorySize":24.0,"memoryBandwidth":300.1,"memoryType":"GDDR6","fp16":30.29,"fp32":30.29,"fp64":0.47,"tensorCores":240,"cuda":"8.9","shaders":7424,"tdp":72},{"name":"Quadro RTX 8000 Passive","architecture":"Turing","generation":"Quadro Turing(Tx000)","memorySize":48.0,"memoryBandwidth":624.0,"memoryType":"GDDR6","fp16":29.86,"fp32":14.93,"fp64":0.47,"tensorCores":576,"cuda":"7.5","shaders":4608,"tdp":260},{"name":"Quadro RTX 6000 Passive","architecture":"Turing","generation":"Quadro Turing(Tx000)","memorySize":24.0,"memoryBandwidth":624.0,"memoryType":"GDDR6","fp16":29.86,"fp32":14.93,"fp64":0.47,"tensorCores":576,"cuda":"7.5","shaders":4608,"tdp":260},{"name":"GeForce RTX 3080","architecture":"Ampere","generation":"GeForce 30","memorySize":10.0,"memoryBandwidth":760.3,"memoryType":"GDDR6X","fp16":29.77,"fp32":29.77,"fp64":0.47,"tensorCores":272,"cuda":"8.6","shaders":8704,"tdp":320},{"name":"GB10","architecture":"Blackwell 2.0","generation":"Server Blackwell(Bxx)","memorySize":128.0,"memoryBandwidth":273.2,"memoryType":"LPDDR5X","fp16":29.71,"fp32":29.71,"fp64":0.46,"tensorCores":384,"cuda":"12.1","shaders":6144,"tdp":140},{"name":"GeForce RTX 4070","architecture":"Ada Lovelace","generation":"GeForce 40","memorySize":12.0,"memoryBandwidth":504.2,"memoryType":"GDDR6X","fp16":29.15,"fp32":29.15,"fp64":0.46,"tensorCores":184,"cuda":"8.9","shaders":5888,"tdp":200},{"name":"GeForce RTX 4070 AD103","architecture":"Ada Lovelace","generation":"GeForce 40","memorySize":12.0,"memoryBandwidth":504.2,"memoryType":"GDDR6X","fp16":29.15,"fp32":29.15,"fp64":0.46,"tensorCores":184,"cuda":"8.9","shaders":5888,"tdp":200},{"name":"GeForce RTX 4070 GDDR6","architecture":"Ada Lovelace","generation":"GeForce 40","memorySize":12.0,"memoryBandwidth":480.0,"memoryType":"GDDR6","fp16":29.15,"fp32":29.15,"fp64":0.46,"tensorCores":184,"cuda":"8.9","shaders":5888,"tdp":200},{"name":"GeForce RTX 4090 Max-Q","architecture":"Ada Lovelace","generation":"GeForce 40 Mobile","memorySize":16.0,"memoryBandwidth":576.0,"memoryType":"GDDR6","fp16":28.31,"fp32":28.31,"fp64":0.44,"tensorCores":304,"cuda":"8.9","shaders":9728,"tdp":80},{"name":"RTX A5000","architecture":"Ampere","generation":"Workstation Ampere(Ax000)","memorySize":24.0,"memoryBandwidth":768.0,"memoryType":"GDDR6","fp16":27.77,"fp32":27.77,"fp64":0.43,"tensorCores":256,"cuda":"8.6","shaders":8192,"tdp":230},{"name":"RTX A5000-12Q","architecture":"Ampere","generation":"Workstation Ampere(Ax000)","memorySize":12.0,"memoryBandwidth":768.0,"memoryType":"GDDR6","fp16":27.77,"fp32":27.77,"fp64":0.43,"tensorCores":256,"cuda":"8.6","shaders":8192,"tdp":230},{"name":"RTX A5000-8Q","architecture":"Ampere","generation":"Workstation Ampere(Ax000)","memorySize":8.0,"memoryBandwidth":768.0,"memoryType":"GDDR6","fp16":27.77,"fp32":27.77,"fp64":0.43,"tensorCores":256,"cuda":"8.6","shaders":8192,"tdp":230},{"name":"GeForce RTX 2080 Ti","architecture":"Turing","generation":"GeForce 20","memorySize":11.0,"memoryBandwidth":616.0,"memoryType":"GDDR6","fp16":26.9,"fp32":13.45,"fp64":0.42,"tensorCores":544,"cuda":"7.5","shaders":4352,"tdp":250},{"name":"Quadro RTX 6000 Mobile","architecture":"Turing","generation":"Quadro Turing-M(Tx000)","memorySize":24.0,"memoryBandwidth":672.0,"memoryType":"GDDR6","fp16":26.82,"fp32":13.41,"fp64":0.42,"tensorCores":576,"cuda":"7.5","shaders":4608},{"name":"RTX 4000 Ada Generation","architecture":"Ada Lovelace","generation":"Workstation Ada(x000A)","memorySize":20.0,"memoryBandwidth":360.0,"memoryType":"GDDR6","fp16":26.73,"fp32":26.73,"fp64":0.42,"tensorCores":192,"cuda":"8.9","shaders":6144,"tdp":130},{"name":"RTX PRO 4000 Blackwell SFF","architecture":"Blackwell 2.0","generation":"Blackwell PRO W(x000)","memorySize":24.0,"memoryBandwidth":432.0,"memoryType":"GDDR7","fp16":25.66,"fp32":25.66,"fp64":0.4,"tensorCores":280,"cuda":"12.0","shaders":8960,"tdp":70},{"name":"GeForce RTX 4080 Mobile","architecture":"Ada Lovelace","generation":"GeForce 40 Mobile","memorySize":12.0,"memoryBandwidth":432.0,"memoryType":"GDDR6","fp16":24.72,"fp32":24.72,"fp64":0.39,"tensorCores":232,"cuda":"8.9","shaders":7424,"tdp":110},{"name":"RTX 4000 Mobile Ada Generation","architecture":"Ada Lovelace","generation":"Ada-MW(x000A)","memorySize":12.0,"memoryBandwidth":432.0,"memoryType":"GDDR6","fp16":24.72,"fp32":24.72,"fp64":0.39,"tensorCores":232,"cuda":"8.9","shaders":7424,"tdp":110},{"name":"GeForce RTX 5060 Ti 16 GB","architecture":"Blackwell 2.0","generation":"GeForce 50","memorySize":16.0,"memoryBandwidth":448.0,"memoryType":"GDDR7","fp16":23.7,"fp32":23.7,"fp64":0.37,"tensorCores":144,"cuda":"12.0","shaders":4608,"tdp":180},{"name":"GeForce RTX 5060 Ti 8 GB","architecture":"Blackwell 2.0","generation":"GeForce 50","memorySize":8.0,"memoryBandwidth":448.0,"memoryType":"GDDR7","fp16":23.7,"fp32":23.7,"fp64":0.37,"tensorCores":144,"cuda":"12.0","shaders":4608,"tdp":180},{"name":"RTX A4500","architecture":"Ampere","generation":"Workstation Ampere(Ax000)","memorySize":20.0,"memoryBandwidth":640.0,"memoryType":"GDDR6","fp16":23.65,"fp32":23.65,"fp64":0.37,"tensorCores":224,"cuda":"8.6","shaders":7168,"tdp":200},{"name":"A10M","architecture":"Ampere","generation":"Server Ampere(Axx)","memorySize":20.0,"memoryBandwidth":500.2,"memoryType":"GDDR6","fp16":23.44,"fp32":23.44,"fp64":0.73,"tensorCores":224,"cuda":"8.6","shaders":7168,"tdp":150},{"name":"GeForce RTX 5080 Mobile","architecture":"Blackwell 2.0","generation":"GeForce 50 Mobile","memorySize":16.0,"memoryBandwidth":896.0,"memoryType":"GDDR7","fp16":23.04,"fp32":23.04,"fp64":0.36,"tensorCores":240,"cuda":"12.0","shaders":7680,"tdp":80},{"name":"RTX 3500 Embedded Ada Generation","architecture":"Ada Lovelace","generation":"Ada-MW(x000A)","memorySize":12.0,"memoryBandwidth":432.0,"memoryType":"GDDR6","fp16":23.04,"fp32":23.04,"fp64":0.36,"tensorCores":160,"cuda":"8.9","shaders":5120,"tdp":100},{"name":"Quadro RTX 5000","architecture":"Turing","generation":"Quadro Turing(Tx000)","memorySize":16.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":22.3,"fp32":11.15,"fp64":0.35,"tensorCores":384,"cuda":"7.5","shaders":3072,"tdp":230},{"name":"GeForce RTX 2080 SUPER","architecture":"Turing","generation":"GeForce 20","memorySize":8.0,"memoryBandwidth":495.9,"memoryType":"GDDR6","fp16":22.3,"fp32":11.15,"fp64":0.35,"tensorCores":384,"cuda":"7.5","shaders":3072,"tdp":250},{"name":"RTX A5500 Mobile","architecture":"Ampere","generation":"Ampere-MW(Ax000)","memorySize":16.0,"memoryBandwidth":512.0,"memoryType":"GDDR6","fp16":22.27,"fp32":22.27,"fp64":0.35,"tensorCores":232,"cuda":"8.6","shaders":7424,"tdp":165},{"name":"CMP 50HX","architecture":"Turing","generation":"Mining GPUs","memorySize":10.0,"memoryBandwidth":560.0,"memoryType":"GDDR6","fp16":22.15,"fp32":11.07,"fp64":0.35,"tensorCores":448,"cuda":"7.5","shaders":3584,"tdp":250},{"name":"GeForce RTX 4060 Ti 16 GB","architecture":"Ada Lovelace","generation":"GeForce 40","memorySize":16.0,"memoryBandwidth":288.0,"memoryType":"GDDR6","fp16":22.06,"fp32":22.06,"fp64":0.34,"tensorCores":136,"cuda":"8.9","shaders":4352,"tdp":165},{"name":"GeForce RTX 4060 Ti 8 GB","architecture":"Ada Lovelace","generation":"GeForce 40","memorySize":8.0,"memoryBandwidth":288.0,"memoryType":"GDDR6","fp16":22.06,"fp32":22.06,"fp64":0.34,"tensorCores":136,"cuda":"8.9","shaders":4352,"tdp":160},{"name":"GeForce RTX 4060 Ti AD104","architecture":"Ada Lovelace","generation":"GeForce 40","memorySize":8.0,"memoryBandwidth":288.0,"memoryType":"GDDR6","fp16":22.06,"fp32":22.06,"fp64":0.34,"tensorCores":136,"cuda":"8.9","shaders":4352,"tdp":160},{"name":"CMP 90HX","architecture":"Ampere","generation":"Mining GPUs","memorySize":10.0,"memoryBandwidth":760.3,"memoryType":"GDDR6X","fp16":21.89,"fp32":21.89,"fp64":0.34,"tensorCores":200,"cuda":"8.6","shaders":6400,"tdp":320},{"name":"GeForce RTX 3070 Ti","architecture":"Ampere","generation":"GeForce 30","memorySize":8.0,"memoryBandwidth":608.3,"memoryType":"GDDR6X","fp16":21.75,"fp32":21.75,"fp64":0.34,"tensorCores":192,"cuda":"8.6","shaders":6144,"tdp":290},{"name":"GeForce RTX 3070 Ti 8 GB GA102","architecture":"Ampere","generation":"GeForce 30","memorySize":8.0,"memoryBandwidth":608.3,"memoryType":"GDDR6X","fp16":21.75,"fp32":21.75,"fp64":0.34,"tensorCores":192,"cuda":"8.6","shaders":6144,"tdp":290},{"name":"GeForce RTX 3070","architecture":"Ampere","generation":"GeForce 30","memorySize":8.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":20.31,"fp32":20.31,"fp64":0.32,"tensorCores":184,"cuda":"8.6","shaders":5888,"tdp":220},{"name":"GeForce RTX 2080","architecture":"Turing","generation":"GeForce 20","memorySize":8.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":20.14,"fp32":10.07,"fp64":0.31,"tensorCores":368,"cuda":"7.5","shaders":2944,"tdp":215},{"name":"GeForce RTX 4080 Max-Q","architecture":"Ada Lovelace","generation":"GeForce 40 Mobile","memorySize":12.0,"memoryBandwidth":432.0,"memoryType":"GDDR6","fp16":20.04,"fp32":20.04,"fp64":0.31,"tensorCores":232,"cuda":"8.9","shaders":7424,"tdp":60},{"name":"Tesla T10 16 GB","architecture":"Turing","generation":"Tesla Turing(Txx)","memorySize":16.0,"memoryBandwidth":403.2,"memoryType":"GDDR6","fp16":20.0,"fp32":10.0,"fp64":0.31,"tensorCores":448,"cuda":"7.5","shaders":3584,"tdp":150},{"name":"RTX A5000 Mobile","architecture":"Ampere","generation":"Ampere-MW(Ax000)","memorySize":16.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":19.35,"fp32":19.35,"fp64":0.3,"tensorCores":192,"cuda":"8.6","shaders":6144,"tdp":150},{"name":"GeForce RTX 5060","architecture":"Blackwell 2.0","generation":"GeForce 50","memorySize":8.0,"memoryBandwidth":448.0,"memoryType":"GDDR7","fp16":19.18,"fp32":19.18,"fp64":0.3,"tensorCores":120,"cuda":"12.0","shaders":3840,"tdp":145},{"name":"RTX 4000 SFF Ada Generation","architecture":"Ada Lovelace","generation":"Workstation Ada(x000A)","memorySize":20.0,"memoryBandwidth":280.0,"memoryType":"GDDR6","fp16":19.17,"fp32":19.17,"fp64":0.3,"tensorCores":192,"cuda":"8.9","shaders":6144,"tdp":70},{"name":"RTX A4000","architecture":"Ampere","generation":"Workstation Ampere(Ax000)","memorySize":16.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":19.17,"fp32":19.17,"fp64":0.3,"tensorCores":192,"cuda":"8.6","shaders":6144,"tdp":140},{"name":"RTX A4000H","architecture":"Ampere","generation":"Workstation Ampere(Ax000)","memorySize":16.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":19.17,"fp32":19.17,"fp64":0.3,"tensorCores":192,"cuda":"8.6","shaders":6144,"tdp":140},{"name":"GeForce RTX 2080 SUPER Mobile","architecture":"Turing","generation":"GeForce 20 Mobile","memorySize":8.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":19.17,"fp32":9.59,"fp64":0.3,"tensorCores":384,"cuda":"7.5","shaders":3072,"tdp":150},{"name":"GeForce RTX 3080 Mobile","architecture":"Ampere","generation":"GeForce 30 Mobile","memorySize":8.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":18.98,"fp32":18.98,"fp64":0.3,"tensorCores":192,"cuda":"8.6","shaders":6144,"tdp":115},{"name":"Quadro RTX 5000 Mobile","architecture":"Turing","generation":"Quadro Turing-M(Tx000)","memorySize":16.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":18.8,"fp32":9.4,"fp64":0.29,"tensorCores":384,"cuda":"7.5","shaders":3072,"tdp":110},{"name":"Quadro RTX 5000 Mobile Refresh","architecture":"Turing","generation":"Quadro Turing-M(Tx000)","memorySize":16.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":18.8,"fp32":9.4,"fp64":0.29,"tensorCores":384,"cuda":"7.5","shaders":3072,"tdp":110},{"name":"Quadro RTX 5000 X2 Mobile","architecture":"Turing","generation":"Quadro Turing-M(Tx000)","memorySize":16.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":18.8,"fp32":9.4,"fp64":0.29,"tensorCores":384,"cuda":"7.5","shaders":3072,"tdp":110},{"name":"GeForce RTX 2080 Mobile","architecture":"Turing","generation":"GeForce 20 Mobile","memorySize":8.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":18.72,"fp32":9.36,"fp64":0.29,"tensorCores":368,"cuda":"7.5","shaders":2944,"tdp":150},{"name":"GeForce RTX 3080 Ti Mobile","architecture":"Ampere","generation":"GeForce 30 Mobile","memorySize":16.0,"memoryBandwidth":512.0,"memoryType":"GDDR6","fp16":18.71,"fp32":18.71,"fp64":0.29,"tensorCores":232,"cuda":"8.6","shaders":7424,"tdp":115},{"name":"RTX A5500 Max-Q","architecture":"Ampere","generation":"Ampere-MW(Ax000)","memorySize":16.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":18.71,"fp32":18.71,"fp64":0.29,"tensorCores":232,"cuda":"8.6","shaders":7424,"tdp":80},{"name":"GeForce RTX 2070 SUPER","architecture":"Turing","generation":"GeForce 20","memorySize":8.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":18.12,"fp32":9.06,"fp64":0.28,"tensorCores":320,"cuda":"7.5","shaders":2560,"tdp":215},{"name":"RTX A4500 Mobile","architecture":"Ampere","generation":"Ampere-MW(Ax000)","memorySize":16.0,"memoryBandwidth":512.0,"memoryType":"GDDR6","fp16":17.66,"fp32":17.66,"fp64":0.28,"tensorCores":184,"cuda":"8.6","shaders":5888,"tdp":140},{"name":"RTX A4000 Mobile","architecture":"Ampere","generation":"Ampere-MW(Ax000)","memorySize":8.0,"memoryBandwidth":384.0,"memoryType":"GDDR6","fp16":17.2,"fp32":17.2,"fp64":0.27,"tensorCores":160,"cuda":"8.6","shaders":5120,"tdp":115},{"name":"GeForce RTX 5070 Ti Mobile","architecture":"Blackwell 2.0","generation":"GeForce 50 Mobile","memorySize":12.0,"memoryBandwidth":672.0,"memoryType":"GDDR7","fp16":17.04,"fp32":17.04,"fp64":0.27,"tensorCores":184,"cuda":"12.0","shaders":5888,"tdp":60},{"name":"RTX PRO 2000 Blackwell","architecture":"Blackwell 2.0","generation":"Blackwell PRO W(x000)","memorySize":16.0,"memoryBandwidth":288.0,"memoryType":"GDDR7","fp16":17.03,"fp32":17.03,"fp64":0.27,"tensorCores":136,"cuda":"12.0","shaders":4352,"tdp":70},{"name":"GeForce RTX 3080 Ti Max-Q","architecture":"Ampere","generation":"GeForce 30 Mobile","memorySize":16.0,"memoryBandwidth":384.0,"memoryType":"GDDR6","fp16":16.7,"fp32":16.7,"fp64":0.26,"tensorCores":232,"cuda":"8.6","shaders":7424,"tdp":80},{"name":"GeForce RTX 3070 Ti Mobile","architecture":"Ampere","generation":"GeForce 30 Mobile","memorySize":8.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":16.6,"fp32":16.6,"fp64":0.26,"tensorCores":184,"cuda":"8.6","shaders":5888,"tdp":115},{"name":"GeForce RTX 3070 TiM","architecture":"Ampere","generation":"GeForce 30","memorySize":8.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":16.6,"fp32":16.6,"fp64":0.26,"tensorCores":184,"cuda":"8.6","shaders":5888,"tdp":220},{"name":"Quadro RTX 5000 Max-Q","architecture":"Turing","generation":"Quadro Turing-M(Tx000)","memorySize":16.0,"memoryBandwidth":384.0,"memoryType":"GDDR6","fp16":16.59,"fp32":8.29,"fp64":0.26,"tensorCores":384,"cuda":"7.5","shaders":3072,"tdp":80},{"name":"RTX A5000 Max-Q","architecture":"Ampere","generation":"Ampere-MW(Ax000)","memorySize":16.0,"memoryBandwidth":384.0,"memoryType":"GDDR6","fp16":16.59,"fp32":16.59,"fp64":0.26,"tensorCores":192,"cuda":"8.6","shaders":6144,"tdp":80},{"name":"GeForce RTX 3060 Ti","architecture":"Ampere","generation":"GeForce 30","memorySize":8.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":16.2,"fp32":16.2,"fp64":0.25,"tensorCores":152,"cuda":"8.6","shaders":4864,"tdp":200},{"name":"GeForce RTX 3060 Ti GA103","architecture":"Ampere","generation":"GeForce 30","memorySize":8.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":16.2,"fp32":16.2,"fp64":0.25,"tensorCores":152,"cuda":"8.6","shaders":4864,"tdp":200},{"name":"GeForce RTX 3060 Ti GDDR6X","architecture":"Ampere","generation":"GeForce 30","memorySize":8.0,"memoryBandwidth":608.3,"memoryType":"GDDR6X","fp16":16.2,"fp32":16.2,"fp64":0.25,"tensorCores":152,"cuda":"8.6","shaders":4864,"tdp":225},{"name":"GeForce RTX 3070 Mobile","architecture":"Ampere","generation":"GeForce 30 Mobile","memorySize":8.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":15.97,"fp32":15.97,"fp64":0.25,"tensorCores":160,"cuda":"8.6","shaders":5120,"tdp":115},{"name":"Quadro RTX 4000 Mobile","architecture":"Turing","generation":"Quadro Turing-M(Tx000)","memorySize":8.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":15.97,"fp32":7.99,"fp64":0.25,"tensorCores":320,"cuda":"7.5","shaders":2560,"tdp":110},{"name":"RTX 3500 Mobile Ada Generation","architecture":"Ada Lovelace","generation":"Ada-MW(x000A)","memorySize":12.0,"memoryBandwidth":432.0,"memoryType":"GDDR6","fp16":15.82,"fp32":15.82,"fp64":0.25,"tensorCores":160,"cuda":"8.9","shaders":5120,"tdp":100},{"name":"GeForce RTX 4070 Mobile","architecture":"Ada Lovelace","generation":"GeForce 40 Mobile","memorySize":8.0,"memoryBandwidth":256.0,"memoryType":"GDDR6","fp16":15.62,"fp32":15.62,"fp64":0.24,"tensorCores":144,"cuda":"8.9","shaders":4608,"tdp":115},{"name":"RTX 3000 Mobile Ada Generation","architecture":"Ada Lovelace","generation":"Ada-MW(x000A)","memorySize":8.0,"memoryBandwidth":256.0,"memoryType":"GDDR6","fp16":15.62,"fp32":15.62,"fp64":0.24,"tensorCores":144,"cuda":"8.9","shaders":4608,"tdp":115},{"name":"GeForce RTX 3080 Max-Q","architecture":"Ampere","generation":"GeForce 30 Mobile","memorySize":8.0,"memoryBandwidth":384.0,"memoryType":"GDDR6","fp16":15.3,"fp32":15.3,"fp64":0.24,"tensorCores":192,"cuda":"8.6","shaders":6144,"tdp":80},{"name":"CMP 40HX","architecture":"Turing","generation":"Mining GPUs","memorySize":8.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":15.21,"fp32":7.6,"fp64":0.24,"tensorCores":288,"cuda":"7.5","shaders":2304,"tdp":185},{"name":"GeForce RTX 4060","architecture":"Ada Lovelace","generation":"GeForce 40","memorySize":8.0,"memoryBandwidth":272.0,"memoryType":"GDDR6","fp16":15.11,"fp32":15.11,"fp64":0.24,"tensorCores":96,"cuda":"8.9","shaders":3072,"tdp":115},{"name":"GeForce RTX 4060 AD106","architecture":"Ada Lovelace","generation":"GeForce 40","memorySize":8.0,"memoryBandwidth":272.0,"memoryType":"GDDR6","fp16":15.11,"fp32":15.11,"fp64":0.24,"tensorCores":96,"cuda":"8.9","shaders":3072,"tdp":115},{"name":"GeForce RTX 2070","architecture":"Turing","generation":"GeForce 20","memorySize":8.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":14.93,"fp32":7.46,"fp64":0.23,"tensorCores":288,"cuda":"7.5","shaders":2304,"tdp":175},{"name":"GeForce RTX 2060 12 GB","architecture":"Turing","generation":"GeForce 20","memorySize":12.0,"memoryBandwidth":336.0,"memoryType":"GDDR6","fp16":14.36,"fp32":7.18,"fp64":0.22,"tensorCores":272,"cuda":"7.5","shaders":2176,"tdp":184},{"name":"GeForce RTX 2060 SUPER","architecture":"Turing","generation":"GeForce 20","memorySize":8.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":14.36,"fp32":7.18,"fp64":0.22,"tensorCores":272,"cuda":"7.5","shaders":2176,"tdp":175},{"name":"RTX A4500 Embedded","architecture":"Ampere","generation":"Ampere-MW(Ax000)","memorySize":16.0,"memoryBandwidth":384.0,"memoryType":"GDDR6","fp16":14.31,"fp32":14.31,"fp64":0.22,"tensorCores":184,"cuda":"8.6","shaders":5888,"tdp":80},{"name":"RTX A4500 Max-Q","architecture":"Ampere","generation":"Ampere-MW(Ax000)","memorySize":16.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":14.31,"fp32":14.31,"fp64":0.22,"tensorCores":184,"cuda":"8.6","shaders":5888,"tdp":80},{"name":"RTX A4000 Max-Q","architecture":"Ampere","generation":"Ampere-MW(Ax000)","memorySize":8.0,"memoryBandwidth":352.0,"memoryType":"GDDR6","fp16":14.28,"fp32":14.28,"fp64":0.22,"tensorCores":160,"cuda":"8.6","shaders":5120,"tdp":80},{"name":"Quadro RTX 4000","architecture":"Turing","generation":"Quadro Turing(Tx000)","memorySize":8.0,"memoryBandwidth":416.0,"memoryType":"GDDR6","fp16":14.24,"fp32":7.12,"fp64":0.22,"tensorCores":288,"cuda":"7.5","shaders":2304,"tdp":160},{"name":"GeForce RTX 2070 SUPER Mobile","architecture":"Turing","generation":"GeForce 20 Mobile","memorySize":8.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":14.13,"fp32":7.07,"fp64":0.22,"tensorCores":320,"cuda":"7.5","shaders":2560,"tdp":115},{"name":"Quadro RTX 4000 Max-Q","architecture":"Turing","generation":"Quadro Turing-M(Tx000)","memorySize":8.0,"memoryBandwidth":384.0,"memoryType":"GDDR6","fp16":14.13,"fp32":7.07,"fp64":0.22,"tensorCores":320,"cuda":"7.5","shaders":2560,"tdp":80},{"name":"GeForce RTX 2070 Mobile Refresh","architecture":"Turing","generation":"GeForce 20 Mobile","memorySize":8.0,"memoryBandwidth":352.0,"memoryType":"GDDR6","fp16":13.41,"fp32":6.71,"fp64":0.21,"tensorCores":288,"cuda":"7.5","shaders":2304,"tdp":115},{"name":"GeForce RTX 2060 SUPER Mobile","architecture":"Turing","generation":"GeForce 20 Mobile","memorySize":8.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":13.32,"fp32":6.66,"fp64":0.21,"tensorCores":272,"cuda":"7.5","shaders":2176,"tdp":175},{"name":"GeForce RTX 2070 Mobile","architecture":"Turing","generation":"GeForce 20 Mobile","memorySize":8.0,"memoryBandwidth":448.0,"memoryType":"GDDR6","fp16":13.27,"fp32":6.64,"fp64":0.21,"tensorCores":288,"cuda":"7.5","shaders":2304,"tdp":115},{"name":"GeForce RTX 3070 Max-Q","architecture":"Ampere","generation":"GeForce 30 Mobile","memorySize":8.0,"memoryBandwidth":384.0,"memoryType":"GDDR6","fp16":13.21,"fp32":13.21,"fp64":0.21,"tensorCores":160,"cuda":"8.6","shaders":5120,"tdp":80},{"name":"GeForce RTX 5050","architecture":"Blackwell 2.0","generation":"GeForce 50","memorySize":8.0,"memoryBandwidth":320.0,"memoryType":"GDDR6","fp16":13.17,"fp32":13.17,"fp64":0.21,"tensorCores":80,"cuda":"12.0","shaders":2560,"tdp":130},{"name":"GeForce RTX 5070 Mobile","architecture":"Blackwell 2.0","generation":"GeForce 50 Mobile","memorySize":8.0,"memoryBandwidth":384.0,"memoryType":"GDDR7","fp16":13.13,"fp32":13.13,"fp64":0.21,"tensorCores":144,"cuda":"12.0","shaders":4608,"tdp":50},{"name":"RTX 2000 Mobile Ada Generation","architecture":"Ada Lovelace","generation":"Ada-MW(x000A)","memorySize":8.0,"memoryBandwidth":256.0,"memoryType":"GDDR6","fp16":12.99,"fp32":12.99,"fp64":0.2,"tensorCores":96,"cuda":"8.9","shaders":3072,"tdp":50},{"name":"GeForce RTX 2080 Max-Q","architecture":"Turing","generation":"GeForce 20 Mobile","memorySize":8.0,"memoryBandwidth":384.0,"memoryType":"GDDR6","fp16":12.89,"fp32":6.45,"fp64":0.2,"tensorCores":368,"cuda":"7.5","shaders":2944,"tdp":80},{"name":"GeForce RTX 3060 12 GB","architecture":"Ampere","generation":"GeForce 30","memorySize":12.0,"memoryBandwidth":360.0,"memoryType":"GDDR6","fp16":12.74,"fp32":12.74,"fp64":0.2,"tensorCores":112,"cuda":"8.6","shaders":3584,"tdp":170},{"name":"GeForce RTX 3060 12 GB GA104","architecture":"Ampere","generation":"GeForce 30","memorySize":12.0,"memoryBandwidth":360.0,"memoryType":"GDDR6","fp16":12.74,"fp32":12.74,"fp64":0.2,"tensorCores":112,"cuda":"8.6","shaders":3584,"tdp":170},{"name":"GeForce RTX 3060 8 GB","architecture":"Ampere","generation":"GeForce 30","memorySize":8.0,"memoryBandwidth":240.0,"memoryType":"GDDR6","fp16":12.74,"fp32":12.74,"fp64":0.2,"tensorCores":112,"cuda":"8.6","shaders":3584,"tdp":170},{"name":"GeForce RTX 3060 8 GB GA104","architecture":"Ampere","generation":"GeForce 30","memorySize":8.0,"memoryBandwidth":240.0,"memoryType":"GDDR6","fp16":12.74,"fp32":12.74,"fp64":0.2,"tensorCores":112,"cuda":"8.6","shaders":3584,"tdp":195},{"name":"RTX 2000 Embedded Ada Generation","architecture":"Ada Lovelace","generation":"Ada-MW(x000A)","memorySize":8.0,"memoryBandwidth":256.0,"memoryType":"GDDR6","fp16":12.35,"fp32":12.35,"fp64":0.19,"tensorCores":96,"cuda":"8.9","shaders":3072,"tdp":50},{"name":"GeForce RTX 3070 Ti Max-Q","architecture":"Ampere","generation":"GeForce 30 Mobile","memorySize":8.0,"memoryBandwidth":384.0,"memoryType":"GDDR6","fp16":12.19,"fp32":12.19,"fp64":0.19,"tensorCores":184,"cuda":"8.6","shaders":5888,"tdp":80},{"name":"RTX 2000 Ada Generation","architecture":"Ada Lovelace","generation":"Workstation Ada(x000A)","memorySize":16.0,"memoryBandwidth":256.0,"memoryType":"GDDR6","fp16":12.0,"fp32":12.0,"fp64":0.19,"tensorCores":88,"cuda":"8.9","shaders":2816,"tdp":70},{"name":"GeForce RTX 2080 SUPER Max-Q","architecture":"Turing","generation":"GeForce 20 Mobile","memorySize":8.0,"memoryBandwidth":352.0,"memoryType":"GDDR6","fp16":11.98,"fp32":5.99,"fp64":0.19,"tensorCores":384,"cuda":"7.5","shaders":3072,"tdp":80},{"name":"GeForce RTX 2070 SUPER Max-Q","architecture":"Turing","generation":"GeForce 20 Mobile","memorySize":8.0,"memoryBandwidth":352.0,"memoryType":"GDDR6","fp16":11.83,"fp32":5.91,"fp64":0.18,"tensorCores":320,"cuda":"7.5","shaders":2560,"tdp":80},{"name":"RTX A3000 Mobile 12 GB","architecture":"Ampere","generation":"Ampere-MW(Ax000)","memorySize":12.0,"memoryBandwidth":336.0,"memoryType":"GDDR6","fp16":11.8,"fp32":11.8,"fp64":0.18,"tensorCores":128,"cuda":"8.6","shaders":4096,"tdp":115},{"name":"GeForce RTX 4060 Mobile","architecture":"Ada Lovelace","generation":"GeForce 40 Mobile","memorySize":8.0,"memoryBandwidth":256.0,"memoryType":"GDDR6","fp16":11.61,"fp32":11.61,"fp64":0.18,"tensorCores":96,"cuda":"8.9","shaders":3072,"tdp":115},{"name":"GeForce RTX 4070 Max-Q","architecture":"Ada Lovelace","generation":"GeForce 40 Mobile","memorySize":8.0,"memoryBandwidth":256.0,"memoryType":"GDDR6","fp16":11.34,"fp32":11.34,"fp64":0.18,"tensorCores":144,"cuda":"8.9","shaders":4608,"tdp":35},{"name":"GeForce RTX 2070 Max-Q","architecture":"Turing","generation":"GeForce 20 Mobile","memorySize":8.0,"memoryBandwidth":384.0,"memoryType":"GDDR6","fp16":10.92,"fp32":5.46,"fp64":0.17,"tensorCores":288,"cuda":"7.5","shaders":2304,"tdp":90},{"name":"CMP 70HX","architecture":"Ampere","generation":"Mining GPUs","memorySize":8.0,"memoryBandwidth":608.3,"memoryType":"GDDR6X","fp16":10.71,"fp32":10.71,"fp64":0.17,"tensorCores":120,"cuda":"8.6","shaders":3840},{"name":"Jetson AGX Orin 64 GB","architecture":"Ampere","generation":"Tegra(Ampere)","memorySize":64.0,"memoryBandwidth":204.8,"memoryType":"LPDDR5","fp16":10.65,"fp32":5.33,"tensorCores":64,"cuda":"8.7","shaders":2048,"tdp":60},{"name":"GeForce RTX 2070 Max-Q Refresh","architecture":"Turing","generation":"GeForce 20 Mobile","memorySize":8.0,"memoryBandwidth":352.0,"memoryType":"GDDR6","fp16":10.37,"fp32":5.18,"fp64":0.16,"tensorCores":288,"cuda":"7.5","shaders":2304,"tdp":115},{"name":"A30 PCIe","architecture":"Ampere","generation":"Server Ampere(Axx)","memorySize":24.0,"memoryBandwidth":933.1,"memoryType":"HBM2e","fp16":10.32,"fp32":10.32,"fp64":5.16,"tensorCores":224,"cuda":"8.0","shaders":3584,"tdp":165},{"name":"A30X","architecture":"Ampere","generation":"Server Ampere(Axx)","memorySize":24.0,"memoryBandwidth":1220.0,"memoryType":"HBM2e","fp16":10.32,"fp32":10.32,"fp64":5.16,"tensorCores":224,"cuda":"8.0","shaders":3584,"tdp":230},{"name":"PG506-207","architecture":"Ampere","generation":"Server Ampere(Axx)","memorySize":24.0,"memoryBandwidth":933.1,"memoryType":"HBM2","fp16":10.32,"fp32":10.32,"fp64":5.16,"tensorCores":224,"cuda":"8.0","shaders":3584,"tdp":165},{"name":"PG506-217","architecture":"Ampere","generation":"Server Ampere(Axx)","memorySize":24.0,"memoryBandwidth":933.1,"memoryType":"HBM2","fp16":10.32,"fp32":10.32,"fp64":5.16,"tensorCores":224,"cuda":"8.0","shaders":3584,"tdp":165},{"name":"PG506-232","architecture":"Ampere","generation":"Server Ampere(Axx)","memorySize":24.0,"memoryBandwidth":933.1,"memoryType":"HBM2","fp16":10.32,"fp32":10.32,"fp64":5.16,"tensorCores":224,"cuda":"8.0","shaders":3584,"tdp":165},{"name":"PG506-242","architecture":"Ampere","generation":"Server Ampere(Axx)","memorySize":24.0,"memoryBandwidth":933.1,"memoryType":"HBM2","fp16":10.32,"fp32":10.32,"fp64":5.16,"tensorCores":224,"cuda":"8.0","shaders":3584,"tdp":165},{"name":"GeForce RTX 5060 Mobile","architecture":"Blackwell 2.0","generation":"GeForce 50 Mobile","memorySize":8.0,"memoryBandwidth":384.0,"memoryType":"GDDR7","fp16":9.68,"fp32":9.68,"fp64":0.15,"tensorCores":104,"cuda":"12.0","shaders":3328,"tdp":45},{"name":"GeForce RTX 3050 8 GB","architecture":"Ampere","generation":"GeForce 30","memorySize":8.0,"memoryBandwidth":224.0,"memoryType":"GDDR6","fp16":9.1,"fp32":9.1,"fp64":0.14,"tensorCores":80,"cuda":"8.6","shaders":2560,"tdp":130},{"name":"GeForce RTX 3050 8 GB GA107","architecture":"Ampere","generation":"GeForce 30","memorySize":8.0,"memoryBandwidth":224.0,"memoryType":"GDDR6","fp16":9.1,"fp32":9.1,"fp64":0.14,"tensorCores":80,"cuda":"8.6","shaders":2560,"tdp":115},{"name":"GeForce RTX 4060 Max-Q","architecture":"Ada Lovelace","generation":"GeForce 40 Mobile","memorySize":8.0,"memoryBandwidth":256.0,"memoryType":"GDDR6","fp16":9.03,"fp32":9.03,"fp64":0.14,"tensorCores":96,"cuda":"8.9","shaders":3072,"tdp":35},{"name":"RTX 2000 Max-Q Ada Generation","architecture":"Ada Lovelace","generation":"Ada-MW(x000A)","memorySize":8.0,"memoryBandwidth":256.0,"memoryType":"GDDR6","fp16":8.94,"fp32":8.94,"fp64":0.14,"tensorCores":96,"cuda":"8.9","shaders":3072,"tdp":35},{"name":"Switch 2 GPU","architecture":"Ampere","generation":"Console GPU(Nintendo)","memorySize":12.0,"memoryBandwidth":102.4,"memoryType":"LPDDR5X","fp16":8.6,"fp32":4.3,"fp64":2.15,"tensorCores":48,"cuda":"8.7","shaders":1536,"tdp":40},{"name":"RTX A2000 Mobile 8 GB","architecture":"Ampere","generation":"Ampere-MW(Ax000)","memorySize":8.0,"memoryBandwidth":224.0,"memoryType":"GDDR6","fp16":8.25,"fp32":8.25,"fp64":0.13,"tensorCores":80,"cuda":"8.6","shaders":2560,"tdp":95},{"name":"GeForce RTX 3050 OEM","architecture":"Ampere","generation":"GeForce 30","memorySize":8.0,"memoryBandwidth":224.0,"memoryType":"GDDR6","fp16":8.09,"fp32":8.09,"fp64":0.13,"tensorCores":72,"cuda":"8.6","shaders":2304,"tdp":130},{"name":"RTX A2000 12 GB","architecture":"Ampere","generation":"Workstation Ampere(Ax000)","memorySize":12.0,"memoryBandwidth":288.0,"memoryType":"GDDR6","fp16":7.99,"fp32":7.99,"fp64":0.12,"tensorCores":104,"cuda":"8.6","shaders":3328,"tdp":70},{"name":"GeForce RTX 5050 Mobile","architecture":"Blackwell 2.0","generation":"GeForce 50 Mobile","memorySize":8.0,"memoryBandwidth":384.0,"memoryType":"GDDR7","fp16":7.68,"fp32":7.68,"fp64":0.12,"tensorCores":80,"cuda":"12.0","shaders":2560,"tdp":50},{"name":"RTX A1000","architecture":"Ampere","generation":"Workstation Ampere(Ax000)","memorySize":8.0,"memoryBandwidth":192.0,"memoryType":"GDDR6","fp16":6.74,"fp32":6.74,"fp64":0.11,"tensorCores":72,"cuda":"8.6","shaders":2304,"tdp":50},{"name":"Jetson AGX Orin 32 GB","architecture":"Ampere","generation":"Tegra(Ampere)","memorySize":32.0,"memoryBandwidth":204.8,"memoryType":"LPDDR5","fp16":6.67,"fp32":3.33,"tensorCores":56,"cuda":"8.7","shaders":1792,"tdp":40},{"name":"RTX A2000 Max-Q 8 GB","architecture":"Ampere","generation":"Ampere-MW(Ax000)","memorySize":8.0,"memoryBandwidth":176.0,"memoryType":"GDDR6","fp16":6.03,"fp32":6.03,"fp64":0.09,"tensorCores":80,"cuda":"8.6","shaders":2560,"tdp":95},{"name":"T1000 8 GB","architecture":"Turing","generation":"Quadro Turing(Tx000)","memorySize":8.0,"memoryBandwidth":160.0,"memoryType":"GDDR6","fp16":5.0,"fp32":2.5,"fp64":0.08,"cuda":"7.5","shaders":896,"tdp":50},{"name":"A2","architecture":"Ampere","generation":"Workstation Ampere(Ax000)","memorySize":16.0,"memoryBandwidth":200.1,"memoryType":"GDDR6","fp16":4.53,"fp32":4.53,"fp64":0.07,"tensorCores":40,"cuda":"8.6","shaders":1280,"tdp":60},{"name":"A2 PCIe","architecture":"Ampere","generation":"Server Ampere(Axx)","memorySize":16.0,"memoryBandwidth":200.1,"memoryType":"GDDR6","fp16":4.53,"fp32":4.53,"fp64":0.14,"tensorCores":40,"cuda":"8.6","shaders":1280,"tdp":60},{"name":"A16 PCIe","architecture":"Ampere","generation":"Server Ampere(Axx)","memorySize":16.0,"memoryBandwidth":200.1,"memoryType":"GDDR6","fp16":4.49,"fp32":4.49,"fp64":0.14,"tensorCores":40,"cuda":"8.6","shaders":1280,"tdp":250},{"name":"Jetson Orin Nano Super","architecture":"Ampere","generation":"Tegra(Ampere)","memorySize":8.0,"memoryBandwidth":102.4,"memoryType":"LPDDR5","fp16":4.18,"fp32":2.09,"tensorCores":32,"cuda":"8.7","shaders":1024,"tdp":25},{"name":"Jetson Orin NX 16 GB","architecture":"Ampere","generation":"Tegra(Ampere)","memorySize":16.0,"memoryBandwidth":102.4,"memoryType":"LPDDR5","fp16":3.76,"fp32":1.88,"tensorCores":32,"cuda":"8.7","shaders":1024,"tdp":25},{"name":"Jetson Orin NX 8 GB","architecture":"Ampere","generation":"Tegra(Ampere)","memorySize":8.0,"memoryBandwidth":102.4,"memoryType":"LPDDR5","fp16":3.13,"fp32":1.57,"tensorCores":32,"cuda":"8.7","shaders":1024,"tdp":20},{"name":"Jetson Orin Nano 8 GB","architecture":"Ampere","generation":"Tegra(Ampere)","memorySize":8.0,"memoryBandwidth":68.22,"memoryType":"LPDDR5","fp16":2.56,"fp32":1.28,"tensorCores":32,"cuda":"8.7","shaders":1024,"tdp":15}] diff --git a/internal/gpudb/gpudb.go b/internal/gpudb/gpudb.go new file mode 100644 index 0000000..0f18f9c --- /dev/null +++ b/internal/gpudb/gpudb.go @@ -0,0 +1,109 @@ +// Package gpudb provides a trimmed NVIDIA GPU specifications index and a GCP +// accelerator catalog. +// +// NVIDIA data is derived from https://huggingface.co/datasets/Jr23xd23/gpu-database +// (Apache-2.0), originally sourced from TechPowerUp via dbgpu / RightNow. +// FP16/FP32 values are TechPowerUp vector TFLOPS, not NVIDIA tensor-core peak. +// +// GCP data is a curated Compute Engine catalog (accelerator types + A/G-series) +// shipped with the CLI; refresh with `runhug gpu update`. +package gpudb + +import ( + _ "embed" + "strings" +) + +//go:embed data/nvidia.json +var nvidiaJSON []byte + +//go:embed data/amd.json +var amdJSON []byte + +// Spec is one GPU row from the vendored NVIDIA/AMD index. +type Spec struct { + Name string `json:"name"` + Vendor string `json:"vendor,omitempty"` // nvidia | amd + Architecture string `json:"architecture,omitempty"` + Generation string `json:"generation,omitempty"` + MemorySize float64 `json:"memorySize"` + MemoryBandwidth float64 `json:"memoryBandwidth,omitempty"` + MemoryType string `json:"memoryType,omitempty"` + FP16 float64 `json:"fp16,omitempty"` + FP32 float64 `json:"fp32,omitempty"` + FP64 float64 `json:"fp64,omitempty"` + TensorCores int `json:"tensorCores,omitempty"` + CUDA string `json:"cuda,omitempty"` + Shaders int `json:"shaders,omitempty"` + TDP int `json:"tdp,omitempty"` +} + +// ByName returns the best exact or fuzzy match for a marketing / catalog name. +func ByName(name string) (Spec, bool) { + specs, err := Load() + if err != nil || len(specs) == 0 { + return Spec{}, false + } + return Match(specs, name) +} + +// Search filters specs whose name/architecture/generation contain query (case-insensitive). +func Search(specs []Spec, query string) []Spec { + q := strings.ToLower(strings.TrimSpace(query)) + if q == "" { + return specs + } + out := make([]Spec, 0, len(specs)) + for _, s := range specs { + hay := strings.ToLower(s.Name + " " + s.Architecture + " " + s.Generation) + if strings.Contains(hay, q) { + out = append(out, s) + } + } + return out +} + +// Match finds the best Spec for a free-form GPU name. +func Match(specs []Spec, name string) (Spec, bool) { + n := normalizeName(name) + if n == "" { + return Spec{}, false + } + var exact Spec + exactOK := false + for _, s := range specs { + if normalizeName(s.Name) == n { + return s, true + } + if !exactOK && strings.EqualFold(s.Name, name) { + exact, exactOK = s, true + } + } + if exactOK { + return exact, true + } + bestScore := 0 + var best Spec + for _, s := range specs { + sn := normalizeName(s.Name) + score := 0 + if sn == n { + return s, true + } + if strings.Contains(sn, n) { + score = len(n)*2 + len(sn) + } else if strings.Contains(n, sn) && len(sn) >= 4 { + score = len(sn)*2 + 1 + } else { + score = tokenOverlap(n, sn) + } + if score > bestScore { + bestScore = score + best = s + } + } + if bestScore >= 6 { + return best, true + } + return Spec{}, false +} diff --git a/internal/gpudb/gpudb_test.go b/internal/gpudb/gpudb_test.go new file mode 100644 index 0000000..6afb6c5 --- /dev/null +++ b/internal/gpudb/gpudb_test.go @@ -0,0 +1,76 @@ +package gpudb_test + +import ( + "strings" + "testing" + + "github.com/adamsiwiec1/runhug/internal/gpudb" +) + +func TestLoad(t *testing.T) { + specs, err := gpudb.Load() + if err != nil { + t.Fatal(err) + } + if len(specs) < 50 { + t.Fatalf("expected trimmed index, got %d", len(specs)) + } +} + +func TestByName(t *testing.T) { + cases := []struct { + in string + want string // substring of matched name + }{ + {"GeForce RTX 4090", "4090"}, + {"RTX 4090", "4090"}, + {"H100 SXM5 80 GB", "H100"}, + {"L4", "L4"}, + {"Tesla T4", "T4"}, + {"L40S", "L40S"}, + {"MI300X", "MI300X"}, + {"Radeon RX 7900 XTX", "7900"}, + } + for _, tc := range cases { + s, ok := gpudb.ByName(tc.in) + if !ok { + t.Fatalf("%q: no match", tc.in) + } + if !strings.Contains(strings.ToUpper(s.Name), strings.ToUpper(tc.want)) { + t.Fatalf("%q -> %q, want contains %q", tc.in, s.Name, tc.want) + } + if tc.in == "L4" && strings.Contains(strings.ToUpper(s.Name), "L40") { + t.Fatalf("L4 matched %q", s.Name) + } + } +} + +func TestLoadIncludesAMD(t *testing.T) { + specs, err := gpudb.Load() + if err != nil { + t.Fatal(err) + } + amd, nv := 0, 0 + for _, s := range specs { + switch s.Vendor { + case "amd": + amd++ + case "nvidia", "": + nv++ + } + } + if amd < 20 { + t.Fatalf("expected AMD rows, got %d (nvidia-ish %d, total %d)", amd, nv, len(specs)) + } +} + +func TestSearch(t *testing.T) { + specs, err := gpudb.Load() + if err != nil { + t.Fatal(err) + } + got := gpudb.Search(specs, "4090") + if len(got) == 0 { + t.Fatal("expected 4090 hits") + } +} diff --git a/internal/gpudb/match.go b/internal/gpudb/match.go new file mode 100644 index 0000000..c7beede --- /dev/null +++ b/internal/gpudb/match.go @@ -0,0 +1,49 @@ +package gpudb + +import ( + "regexp" + "strings" +) + +var nonAlnum = regexp.MustCompile(`[^a-z0-9]+`) + +func normalizeName(s string) string { + s = strings.ToLower(strings.TrimSpace(s)) + for _, p := range []string{"nvidia ", "geforce ", "tesla ", "quadro "} { + s = strings.ReplaceAll(s, p, "") + } + return nonAlnum.ReplaceAllString(s, "") +} + +func tokenOverlap(a, b string) int { + if len(a) < 3 || len(b) < 3 { + return 0 + } + score := 0 + for _, tok := range distinctiveTokens(a) { + if len(tok) >= 3 && strings.Contains(b, tok) { + score += len(tok) + } + } + return score +} + +func distinctiveTokens(n string) []string { + var out []string + var cur strings.Builder + flush := func() { + if cur.Len() >= 2 { + out = append(out, cur.String()) + } + cur.Reset() + } + for _, r := range n { + if (r >= '0' && r <= '9') || (r >= 'a' && r <= 'z') { + cur.WriteRune(r) + } else { + flush() + } + } + flush() + return out +} diff --git a/internal/gpudb/update.go b/internal/gpudb/update.go new file mode 100644 index 0000000..149ab98 --- /dev/null +++ b/internal/gpudb/update.go @@ -0,0 +1,303 @@ +package gpudb + +import ( + "context" + "encoding/json" + "fmt" + "io" + "net/http" + "os" + "os/exec" + "sort" + "strings" + "time" + + "github.com/adamsiwiec1/runhug/internal/packs" + "github.com/adamsiwiec1/runhug/internal/version" +) + +const ( + releaseAssetNVIDIA = "gpudb-nvidia.json" + releaseAssetAMD = "gpudb-amd.json" + releaseAssetGCP = "gpudb-gcp.json" +) + +// UpdateOptions controls gpu update. +type UpdateOptions struct { + NVIDIA bool + AMD bool + GCP bool + // PreferRelease tries GitHub Release assets first (like model packs). + PreferRelease bool +} + +// UpdateResult summarizes what was written. +type UpdateResult struct { + NVIDIACount int + AMDCount int + GCPCount int + NVIDIAFrom string + AMDFrom string + GCPFrom string + CacheDir string +} + +// Update refreshes cached catalogs under CacheDir. +func Update(ctx context.Context, opts UpdateOptions) (UpdateResult, error) { + if !opts.NVIDIA && !opts.AMD && !opts.GCP { + opts.NVIDIA, opts.AMD, opts.GCP = true, true, true + } + dir, err := CacheDir() + if err != nil { + return UpdateResult{}, err + } + res := UpdateResult{CacheDir: dir} + + if opts.NVIDIA { + n, src, err := updateVendor(ctx, updateVendorOpts{ + PreferRelease: opts.PreferRelease, + ReleaseAsset: releaseAssetNVIDIA, + FetchURL: NVIDIAFetchURL, + CacheName: nvidiaCache, + Vendor: "nvidia", + Trim: TrimNVIDIA, + Label: "huggingface:Jr23xd23/gpu-database/nvidia", + }) + if err != nil { + return res, err + } + res.NVIDIACount, res.NVIDIAFrom = n, src + } + if opts.AMD { + n, src, err := updateVendor(ctx, updateVendorOpts{ + PreferRelease: opts.PreferRelease, + ReleaseAsset: releaseAssetAMD, + FetchURL: AMDFetchURL, + CacheName: amdCache, + Vendor: "amd", + Trim: TrimAMD, + Label: "huggingface:Jr23xd23/gpu-database/amd", + }) + if err != nil { + return res, err + } + res.AMDCount, res.AMDFrom = n, src + } + if opts.GCP { + n, src, err := updateGCP(ctx, opts.PreferRelease) + if err != nil { + return res, err + } + res.GCPCount, res.GCPFrom = n, src + } + return res, nil +} + +type updateVendorOpts struct { + PreferRelease bool + ReleaseAsset string + FetchURL string + CacheName string + Vendor string + Trim func([]Spec) []Spec + Label string +} + +func updateVendor(ctx context.Context, o updateVendorOpts) (int, string, error) { + var raw []byte + var src string + if o.PreferRelease { + if b, err := fetchReleaseAsset(ctx, o.ReleaseAsset); err == nil && len(b) > 0 { + raw, src = b, "github-release:"+o.ReleaseAsset + } + } + if raw == nil { + b, err := httpGet(ctx, o.FetchURL) + if err != nil { + return 0, "", fmt.Errorf("fetch %s index: %w", o.Vendor, err) + } + raw, src = b, o.Label + } + var full []Spec + if err := json.Unmarshal(raw, &full); err != nil { + return 0, "", fmt.Errorf("decode %s JSON: %w", o.Vendor, err) + } + trimmed := o.Trim(full) + if len(trimmed) == 0 { + trimmed = full + for i := range trimmed { + if trimmed[i].Vendor == "" { + trimmed[i].Vendor = o.Vendor + } + } + } + sort.Slice(trimmed, func(i, j int) bool { + if trimmed[i].FP16 != trimmed[j].FP16 { + return trimmed[i].FP16 > trimmed[j].FP16 + } + return trimmed[i].Name < trimmed[j].Name + }) + out, err := json.Marshal(trimmed) + if err != nil { + return 0, "", err + } + out = append(out, '\n') + if err := WriteCache(o.CacheName, out); err != nil { + return 0, "", err + } + return len(trimmed), src, nil +} + +func updateGCP(ctx context.Context, preferRelease bool) (int, string, error) { + var raw []byte + var src string + + // 1) Live gcloud list merged onto curated metadata (best when authenticated). + if live, err := fetchGCloudAccelerators(ctx); err == nil && len(live) > 0 { + merged := mergeGCPLive(live) + out, err := json.Marshal(merged) + if err != nil { + return 0, "", err + } + out = append(out, '\n') + if err := WriteCache(gcpCache, out); err != nil { + return 0, "", err + } + return len(merged), "gcloud+bundled", nil + } + + if preferRelease { + if b, err := fetchReleaseAsset(ctx, releaseAssetGCP); err == nil && len(b) > 0 { + raw, src = b, "github-release:"+releaseAssetGCP + } + } + if raw == nil { + b, err := httpGet(ctx, GCPCatalogURL) + if err != nil { + // Fall back to re-writing the embed so cache exists for inspection. + raw = append([]byte(nil), gcpJSON...) + src = "bundled" + } else { + raw, src = b, "github:openhat-security/runhug/gcp.json" + } + } + var list []GCPAccelerator + if err := json.Unmarshal(raw, &list); err != nil { + return 0, "", fmt.Errorf("decode GCP JSON: %w", err) + } + out, err := json.Marshal(list) + if err != nil { + return 0, "", err + } + out = append(out, '\n') + if err := WriteCache(gcpCache, out); err != nil { + return 0, "", err + } + return len(list), src, nil +} + +func fetchReleaseAsset(ctx context.Context, name string) ([]byte, error) { + rc := packs.NewReleaseClient(packs.ReleaseRepo()) + if tok := strings.TrimSpace(os.Getenv("GITHUB_TOKEN")); tok != "" { + rc.Token = tok + } + data, _, err := rc.FetchAssetBytes(ctx, name) + return data, err +} + +func httpGet(ctx context.Context, url string) ([]byte, error) { + req, err := http.NewRequestWithContext(ctx, http.MethodGet, url, nil) + if err != nil { + return nil, err + } + req.Header.Set("User-Agent", version.Name+"/"+version.Version) + client := &http.Client{Timeout: 2 * time.Minute} + res, err := client.Do(req) + if err != nil { + return nil, err + } + defer res.Body.Close() + body, err := io.ReadAll(io.LimitReader(res.Body, 32<<20)) + if err != nil { + return nil, err + } + if res.StatusCode >= 300 { + return nil, fmt.Errorf("HTTP %d from %s", res.StatusCode, url) + } + return body, nil +} + +func fetchGCloudAccelerators(ctx context.Context) ([]string, error) { + path, err := exec.LookPath("gcloud") + if err != nil { + return nil, err + } + cmd := exec.CommandContext(ctx, path, "compute", "accelerator-types", "list", + "--format=value(name)") + out, err := cmd.Output() + if err != nil { + return nil, err + } + seen := map[string]bool{} + var ids []string + for _, line := range strings.Split(string(out), "\n") { + id := strings.TrimSpace(line) + if id == "" || seen[id] { + continue + } + seen[id] = true + ids = append(ids, id) + } + sort.Strings(ids) + return ids, nil +} + +func mergeGCPLive(ids []string) []GCPAccelerator { + curated, _ := LoadGCP() + byID := map[string]GCPAccelerator{} + for _, c := range curated { + byID[c.ID] = c + } + // Always include curated machine-series entries even if not in accelerator-types. + out := make([]GCPAccelerator, 0, len(ids)+len(curated)) + seen := map[string]bool{} + for _, id := range ids { + if !strings.HasPrefix(id, "nvidia-") { + continue // skip TPUs etc. + } + if c, ok := byID[id]; ok { + out = append(out, c) + } else { + out = append(out, GCPAccelerator{ + ID: id, + Name: prettyAccelName(id), + HFName: prettyAccelName(id), + Series: "N1", + Attach: "accelerator", + }) + } + seen[id] = true + } + for _, c := range curated { + if !seen[c.ID] { + out = append(out, c) + } + } + sort.Slice(out, func(i, j int) bool { + if out[i].MemoryGB != out[j].MemoryGB { + return out[i].MemoryGB > out[j].MemoryGB + } + return out[i].Name < out[j].Name + }) + return out +} + +func prettyAccelName(id string) string { + s := strings.TrimPrefix(id, "nvidia-") + s = strings.TrimPrefix(s, "tesla-") + s = strings.ReplaceAll(s, "-", " ") + if s == "" { + return id + } + return strings.ToUpper(s[:1]) + s[1:] +} diff --git a/internal/hostgpu/hostgpu.go b/internal/hostgpu/hostgpu.go new file mode 100644 index 0000000..bdff60d --- /dev/null +++ b/internal/hostgpu/hostgpu.go @@ -0,0 +1,256 @@ +// Package hostgpu probes the machine for discrete GPUs (NVIDIA) or Apple Silicon. +package hostgpu + +import ( + "os/exec" + "regexp" + "runtime" + "strconv" + "strings" + + "github.com/adamsiwiec1/runhug/internal/recommend" +) + +// Device is one detected accelerator on this host. +type Device struct { + Name string // marketing name (e.g. "GeForce RTX 4090", "Apple M3 Max") + MemoryGB float64 // VRAM or unified memory (Mac) + Vendor string // nvidia | apple | unknown + Source string // nvidia-smi | apple-silicon + FP16 float64 // estimated vector FP16 TFLOPS when known (0 = unknown) + GPUCores int // Apple GPU core count when known + Note string // e.g. "FP16≈2×FP32 (public estimates)" +} + +// Detect returns local GPUs. Empty slice means none found (not an error). +func Detect() ([]Device, error) { + if runtime.GOOS == "darwin" && runtime.GOARCH == "arm64" { + return []Device{detectApple()}, nil + } + var out []Device + if nv, err := nvidiaSMI(); err == nil { + out = append(out, nv...) + } + if amd, err := rocmSMI(); err == nil { + out = append(out, amd...) + } + return out, nil +} + +func detectApple() Device { + ram := recommend.RAMGB() + chip := appleChipName() + cores := appleGPUCores() + fp16, note := appleFP16(chip, cores) + name := chip + if name == "" { + name = "Apple Silicon" + } + return Device{ + Name: name, + MemoryGB: ram, + Vendor: "apple", + Source: "apple-silicon", + FP16: fp16, + GPUCores: cores, + Note: note, + } +} + +func appleChipName() string { + out, err := exec.Command("sysctl", "-n", "machdep.cpu.brand_string").Output() + if err == nil { + s := strings.TrimSpace(string(out)) + if strings.HasPrefix(s, "Apple ") { + return s + } + } + // Fallback: system_profiler Chip line + out, err = exec.Command("system_profiler", "SPHardwareDataType").Output() + if err != nil { + return "" + } + for _, line := range strings.Split(string(out), "\n") { + line = strings.TrimSpace(line) + if strings.HasPrefix(line, "Chip:") { + return strings.TrimSpace(strings.TrimPrefix(line, "Chip:")) + } + } + return "" +} + +var coresRe = regexp.MustCompile(`(?i)Total Number of Cores:\s*(\d+)`) + +func appleGPUCores() int { + out, err := exec.Command("system_profiler", "SPDisplaysDataType").Output() + if err != nil { + return 0 + } + // Prefer the first "Total Number of Cores" under the Apple GPU block. + m := coresRe.FindStringSubmatch(string(out)) + if len(m) == 2 { + n, _ := strconv.Atoi(m[1]) + return n + } + return 0 +} + +// appleFP16 returns estimated GPU FP16 TFLOPS for ranking against TechPowerUp vector FP16. +// Sources: Flopper.io FP32 (and M4 FP16), JD Hodges / public charts using FP16≈2×FP32. +// Not Neural Engine TOPS. Values are approximate peaks for ranking only. +func appleFP16(chip string, cores int) (float64, string) { + key := normalizeAppleChip(chip) + spec, ok := appleBaseFP16[key] + if !ok { + return 0, "FP16 unknown for this Apple chip" + } + fp16 := spec.fp16 + if cores > 0 && spec.maxCores > 0 && cores < spec.maxCores { + fp16 = spec.fp16 * float64(cores) / float64(spec.maxCores) + } + return fp16, "FP16 est. ≈2× public FP32 (Flopper/charts); ranking only" +} + +type appleSpec struct { + fp16 float64 + maxCores int +} + +// Peak configs; lower core bins are scaled by detected GPU core count. +var appleBaseFP16 = map[string]appleSpec{ + "apple m1": {5.2, 8}, + "apple m1 pro": {10.4, 16}, + "apple m1 max": {21.2, 32}, + "apple m1 ultra": {42.4, 64}, + "apple m2": {7.2, 10}, + "apple m2 pro": {13.6, 19}, + "apple m2 max": {27.2, 38}, + "apple m2 ultra": {54.4, 76}, + "apple m3": {7.1, 10}, + "apple m3 pro": {14.2, 18}, + "apple m3 max": {28.4, 40}, // 40-core peak; 30-core MacBook Pro scales down + "apple m3 ultra": {56.8, 80}, + "apple m4": {8.5, 10}, // Flopper FP16 + "apple m4 pro": {18.4, 20}, + "apple m4 max": {36.8, 40}, + "apple m4 ultra": {73.6, 80}, + "apple m5": {9.0, 10}, + "apple m5 pro": {20.0, 20}, + "apple m5 max": {33.2, 40}, +} + +func normalizeAppleChip(s string) string { + s = strings.ToLower(strings.TrimSpace(s)) + s = strings.ReplaceAll(s, " ", " ") + return s +} + +func nvidiaSMI() ([]Device, error) { + path, err := exec.LookPath("nvidia-smi") + if err != nil { + return nil, nil + } + out, err := exec.Command(path, "--query-gpu=name,memory.total", "--format=csv,noheader,nounits").CombinedOutput() + if err != nil { + return nil, nil + } + var devices []Device + for _, line := range strings.Split(string(out), "\n") { + line = strings.TrimSpace(line) + if line == "" { + continue + } + parts := strings.SplitN(line, ",", 2) + if len(parts) < 2 { + continue + } + name := strings.TrimSpace(parts[0]) + memStr := strings.TrimSpace(parts[1]) + mb, _ := strconv.ParseFloat(memStr, 64) + gb := mb / 1024 + if gb <= 0 { + continue + } + devices = append(devices, Device{ + Name: name, + MemoryGB: gb, + Vendor: "nvidia", + Source: "nvidia-smi", + }) + } + return devices, nil +} + +// rocmSMI parses `rocm-smi --showproductname --showmeminfo vram` when present. +func rocmSMI() ([]Device, error) { + path, err := exec.LookPath("rocm-smi") + if err != nil { + return nil, nil + } + out, err := exec.Command(path, "--showproductname", "--showmeminfo", "vram", "--csv").CombinedOutput() + if err != nil { + // Fallback: plain text product name only + out2, err2 := exec.Command(path, "--showproductname").CombinedOutput() + if err2 != nil { + return nil, nil + } + name := parseROCmProductName(string(out2)) + if name == "" { + return nil, nil + } + return []Device{{Name: name, Vendor: "amd", Source: "rocm-smi"}}, nil + } + return parseROCmCSV(string(out)), nil +} + +func parseROCmProductName(s string) string { + for _, line := range strings.Split(s, "\n") { + line = strings.TrimSpace(line) + low := strings.ToLower(line) + if strings.Contains(low, "card series") || strings.Contains(low, "card model") || strings.Contains(low, "device name") { + if i := strings.Index(line, ":"); i >= 0 { + return strings.TrimSpace(line[i+1:]) + } + } + } + return "" +} + +func parseROCmCSV(s string) []Device { + // Best-effort: look for lines with a card name and a memory number. + var devices []Device + name := "" + for _, line := range strings.Split(s, "\n") { + line = strings.TrimSpace(line) + if line == "" || strings.HasPrefix(strings.ToLower(line), "device") { + continue + } + parts := strings.Split(line, ",") + for _, p := range parts { + p = strings.TrimSpace(p) + low := strings.ToLower(p) + if strings.Contains(low, "radeon") || strings.Contains(low, "instinct") || strings.Contains(low, "amd") { + name = p + } + } + if name == "" { + continue + } + memGB := 0.0 + for _, p := range parts { + p = strings.TrimSpace(strings.TrimSuffix(strings.ToLower(p), " mb")) + p = strings.TrimSuffix(p, "miB") + if n, err := strconv.ParseFloat(strings.TrimSpace(p), 64); err == nil && n > 256 { + memGB = n / 1024 + } + } + devices = append(devices, Device{ + Name: name, + MemoryGB: memGB, + Vendor: "amd", + Source: "rocm-smi", + }) + name = "" + } + return devices +} diff --git a/internal/hostgpu/hostgpu_test.go b/internal/hostgpu/hostgpu_test.go new file mode 100644 index 0000000..c06b401 --- /dev/null +++ b/internal/hostgpu/hostgpu_test.go @@ -0,0 +1,32 @@ +package hostgpu_test + +import ( + "runtime" + "testing" + + "github.com/adamsiwiec1/runhug/internal/hostgpu" +) + +func TestDetectAppleHasName(t *testing.T) { + if runtime.GOOS != "darwin" || runtime.GOARCH != "arm64" { + t.Skip("Apple Silicon only") + } + devs, err := hostgpu.Detect() + if err != nil { + t.Fatal(err) + } + if len(devs) != 1 { + t.Fatalf("got %d devices", len(devs)) + } + d := devs[0] + if d.Vendor != "apple" { + t.Fatalf("vendor=%s", d.Vendor) + } + if d.Name == "" || d.Name == "Apple Silicon" { + // brand_string should resolve on real Macs + t.Logf("chip name=%q (may be generic in CI)", d.Name) + } + if d.FP16 <= 0 && d.Name != "Apple Silicon" { + t.Fatalf("expected FP16 estimate for %q", d.Name) + } +} diff --git a/internal/packs/download.go b/internal/packs/download.go index 7545b04..d3b592f 100644 --- a/internal/packs/download.go +++ b/internal/packs/download.go @@ -163,6 +163,24 @@ func (c *ReleaseClient) TryDownloadDelta(ctx context.Context, id, destPath strin return true, nil } +// FetchAssetBytes downloads a named asset from the latest release. +// Returns (data, tag, err). Missing asset → error mentioning the name. +func (c *ReleaseClient) FetchAssetBytes(ctx context.Context, name string) ([]byte, string, error) { + rel, err := c.fetchLatest(ctx) + if err != nil { + return nil, "", err + } + url, err := assetURL(rel, name) + if err != nil { + return nil, "", err + } + data, err := c.downloadBytes(ctx, url) + if err != nil { + return nil, "", err + } + return data, rel.TagName, nil +} + func assetURL(rel *ghRelease, name string) (string, error) { for _, a := range rel.Assets { if a.Name == name { diff --git a/scripts/refresh-gpudb.sh b/scripts/refresh-gpudb.sh new file mode 100755 index 0000000..5026303 --- /dev/null +++ b/scripts/refresh-gpudb.sh @@ -0,0 +1,73 @@ +#!/usr/bin/env bash +# Refresh vendored GPU catalogs from Jr23xd23/gpu-database (Apache-2.0). +# Source: TechPowerUp via dbgpu / RightNow. Do not use resolve-cache URLs. +set -euo pipefail +ROOT="$(cd "$(dirname "$0")/.." && pwd)" +trim_vendor() { + local url="$1" out="$2" vendor="$3" + local tmp + tmp="$(mktemp)" + echo "fetching $url" + curl -fsSL "$url" -o "$tmp" + python3 - "$tmp" "$out" "$vendor" <<'PY' +import json, sys +src, dest, vendor = sys.argv[1], sys.argv[2], sys.argv[3] +data = json.load(open(src)) +out = [] +for g in data: + mem = g.get("memorySize") or 0 + if mem < 8: + continue + name = g.get("name") or "" + if vendor == "nvidia": + try: + cuda = float(g["cuda"]) if g.get("cuda") is not None else 0.0 + except (TypeError, ValueError): + cuda = 0.0 + if cuda < 7.5: + continue + else: + low = name.lower() + keep = bool(g.get("fp16") or g.get("fp32")) + if any(p in low for p in ("instinct", "radeon rx", "radeon pro", "radeon vii", "firepro")): + keep = True + if not keep: + continue + row = { + "name": name, + "vendor": vendor, + "architecture": g.get("architecture"), + "generation": g.get("generation"), + "memorySize": mem, + "memoryBandwidth": g.get("memoryBandwidth"), + "memoryType": g.get("memoryType"), + "fp16": g.get("fp16"), + "fp32": g.get("fp32"), + "fp64": g.get("fp64"), + "tensorCores": g.get("tensorCores"), + "cuda": str(g["cuda"]) if g.get("cuda") is not None else "", + "shaders": g.get("shaders"), + "tdp": g.get("tdp"), + } + row = {k: v for k, v in row.items() if v is not None and v != ""} + out.append(row) +out.sort(key=lambda r: (-(r.get("fp16") or 0), -(r.get("memorySize") or 0), r.get("name") or "")) +with open(dest, "w") as f: + json.dump(out, f, separators=(",", ":")) + f.write("\n") +print(f"wrote {len(out)} {vendor} rows -> {dest}") +PY + rm -f "$tmp" +} + +trim_vendor \ + "https://huggingface.co/datasets/Jr23xd23/gpu-database/resolve/main/data/nvidia/all.json" \ + "$ROOT/internal/gpudb/data/nvidia.json" \ + nvidia + +trim_vendor \ + "https://huggingface.co/datasets/Jr23xd23/gpu-database/resolve/main/data/amd/all.json" \ + "$ROOT/internal/gpudb/data/amd.json" \ + amd + +echo "GCP catalog: edit internal/gpudb/data/gcp.json (or runhug gpu update with gcloud auth)"