{
  "count": 5,
  "kind": "all",
  "sorts": [
    "kind",
    "stars",
    "name"
  ],
  "facets": {},
  "facetCounts": {},
  "outputs": [
    "json",
    "xml",
    "csv",
    "sse",
    "events"
  ],
  "resources": [
    {
      "slug": "tensorrt-llm",
      "name": "TensorRT-LLM",
      "kind": "ai-library",
      "tags": [],
      "links": [],
      "url": "https://github.com/NVIDIA/TensorRT-LLM",
      "description": "NVIDIA's optimized LLM inference.",
      "access": "oss",
      "pricing": "free",
      "github": {
        "repo": "NVIDIA/TensorRT-LLM"
      },
      "caps": [
        "inference",
        "gpu"
      ],
      "facets": {
        "lang": "cpp",
        "libType": [
          "inference"
        ],
        "oss": [
          "true"
        ]
      },
      "kinds": [
        "ai-library"
      ]
    },
    {
      "slug": "exllamav2",
      "name": "ExLlamaV2",
      "kind": "ai-library",
      "tags": [],
      "links": [],
      "url": "https://github.com/turboderp-org/exllamav2",
      "description": "Fast local inference kernels.",
      "access": "oss",
      "pricing": "free",
      "github": {
        "repo": "turboderp-org/exllamav2"
      },
      "caps": [
        "inference",
        "gpu"
      ],
      "facets": {
        "lang": "python",
        "libType": [
          "inference"
        ],
        "oss": [
          "true"
        ]
      },
      "kinds": [
        "ai-library"
      ]
    },
    {
      "slug": "modal",
      "name": "Modal",
      "kind": "build-platform",
      "tags": [],
      "links": [],
      "url": "https://modal.com",
      "description": "Serverless GPU/CPU compute.",
      "access": "platform",
      "pricing": "paid",
      "freeTier": true,
      "caps": [
        "gpu",
        "serverless"
      ],
      "kinds": [
        "build-platform"
      ],
      "facets": {
        "tier": [
          "gpu",
          "serverless"
        ],
        "freeTier": [
          "true"
        ]
      }
    },
    {
      "slug": "novita",
      "name": "Novita AI",
      "kind": "model-provider",
      "tags": [
        "gpu"
      ],
      "links": [],
      "url": "https://novita.ai",
      "description": "Serverless GPU model APIs.",
      "access": "api",
      "pricing": "paid",
      "freeTier": true,
      "caps": [
        "text",
        "image-gen",
        "video-gen"
      ],
      "kinds": [
        "model-provider"
      ],
      "facets": {
        "freeTier": [
          "true"
        ],
        "pricing": [
          "paid"
        ],
        "capability": [
          "text",
          "image-gen",
          "video-gen"
        ]
      }
    },
    {
      "slug": "vllm",
      "name": "vLLM",
      "kind": "model-runner",
      "tags": [
        "serving",
        "gpu"
      ],
      "links": [],
      "url": "https://github.com/vllm-project/vllm",
      "description": "High-throughput GPU serving for open models.",
      "access": "self-host",
      "pricing": "free",
      "caps": [
        "server",
        "api"
      ],
      "github": {
        "repo": "vllm-project/vllm",
        "stars": 92134,
        "starsAsOf": "2026-09-19"
      },
      "kinds": [
        "model-runner"
      ],
      "facets": {
        "server": [
          "true"
        ],
        "gpu": [
          "true"
        ]
      }
    }
  ]
}