-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathpyproject.toml
More file actions
87 lines (78 loc) · 3.31 KB
/
Copy pathpyproject.toml
File metadata and controls
87 lines (78 loc) · 3.31 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
[project]
name = "inferhost"
version = "0.14.2"
description = "Self-hosted, multi-modal AI server for your own GPU: chat/vision LLMs (Qwen, Llama, Gemma, DeepSeek), text-to-speech (Kokoro, OuteTTS), and image generation (SDXL, Flux, Z-Image, Qwen-Image) behind one OpenAI-compatible endpoint. Wraps llama.cpp and stable-diffusion.cpp, auto-downloads GGUF models from Hugging Face, hot-swaps VRAM, speculative decoding — no compiling. Type `inferhost` and you're done."
readme = "README.md"
license = { file = "LICENSE" }
requires-python = ">=3.11"
authors = [
{ name = "amirrouh", email = "43451887+amirrouh@users.noreply.github.com" },
]
keywords = [
"llm", "local-llm", "inference-server", "self-hosted", "local-ai",
"openai-compatible", "openai-api", "gguf", "llama-cpp", "llama.cpp",
"huggingface", "quantization", "speculative-decoding", "multimodal",
"vision", "text-to-speech", "tts", "kokoro", "image-generation", "stable-diffusion",
"sdxl", "flux", "qwen", "llama", "gemma", "gpu", "vulkan", "metal",
"ai-server", "model-server", "tui",
]
classifiers = [
"Development Status :: 3 - Alpha",
"Environment :: Console",
"Intended Audience :: Developers",
"License :: OSI Approved :: Apache Software License",
"Operating System :: POSIX :: Linux",
"Operating System :: MacOS :: MacOS X",
"Programming Language :: Python :: 3",
"Programming Language :: Python :: 3.11",
"Programming Language :: Python :: 3.12",
"Programming Language :: Python :: 3.13",
"Topic :: Scientific/Engineering :: Artificial Intelligence",
]
dependencies = [
"textual>=0.80",
"rich>=13",
"huggingface-hub>=0.24",
"httpx>=0.27",
"litellm[proxy]>=1.40",
# litellm's proxy imports fastapi.dependencies.utils.get_flat_dependant, which
# FastAPI dropped in 0.140.7 (and every release after). litellm only asks for
# fastapi<1.0, so a fresh install resolves past that and the gateway dies on
# import — llama-swap keeps serving while port 4000 is refused. Last good
# version is 0.140.6; unpin once litellm stops using the symbol.
"fastapi<0.140.7",
"pydantic-settings>=2",
"psutil>=5.9",
"pyyaml>=6",
"tomli-w>=1",
# Kokoro TTS engine (ONNX Runtime, CPU). Upstream caps python at <3.14, so
# the marker keeps inferhost installable on newer pythons — Kokoro serving
# raises a clear error there instead.
"kokoro-onnx>=0.4.9; python_version < '3.14'",
]
[project.optional-dependencies]
# Deprecated alias: litellm is now a core dependency. Kept so 'inferhost[gateway]' still resolves.
gateway = []
dev = ["pytest>=8", "pytest-asyncio>=0.23", "ruff>=0.5", "mypy>=1.10"]
[project.scripts]
inferhost = "inferhost.cli:app"
[project.urls]
Homepage = "https://github.com/amirrouh/inferhost"
Issues = "https://github.com/amirrouh/inferhost/issues"
[build-system]
# Pin hatchling: newer releases emit Metadata-Version 2.5, which twine (in the
# PyPI publish action) rejects as "not a valid metadata version" and fails the
# release. 1.27.0 emits the widely-accepted 2.4. Bump only when the publish
# toolchain accepts 2.5.
requires = ["hatchling==1.27.0"]
build-backend = "hatchling.build"
[tool.hatch.build.targets.wheel]
packages = ["src/inferhost"]
[tool.ruff]
line-length = 100
target-version = "py311"
[tool.ruff.lint]
select = ["E", "F", "I", "UP", "B", "SIM"]
ignore = ["E501"]
[tool.pytest.ini_options]
testpaths = ["tests"]