swactor/crates/mvp-system/Cargo.toml
Zachery Aaron Shores-Chmielewski 6608824cb0 feat(mvp-chat): local e2e chat on cuda gpu
Stand up an interactive end-to-end chat over a CUDA GPU, provisioning a Dockerized node that loads a GGUF model and serves prompts over TCP.

- prompt_rpc: add the newline-JSON prompt protocol (`SubmitPrompt` + `PromptEvent::{TextDelta,Done,Fault}`) carried over TCP
- mvp_chat: add an interactive REPL client connecting to the prompt RPC port (default 127.0.0.1:19777)
- mvp_orch_one_node / mvp_one_node_chat: add the single-node orchestrator that provisions a `LocalDockerPlugin` node, loads `bartowski/Llama-3.2-1B-Instruct-GGUF` (Q4_K_M), and exposes the prompt RPC listener with boot/route/weight timeouts
- mvp_node: add the GPU worker binary that spawns `tinygrad_worker.py` (default device CUDA) and ships runtime telemetry via a `ClusterFrameSink`
- vastai_provisioning / bootstrap_datastream: add the vast.ai provider adapter (`VastAiProvisioningConfig`, `VastAiLeaseClient`) wrapping `swactor_vastai`, plus a bridge that folds provision stdout onto a per-node datastream
- apps/mvp-node: add CUDA base/runtime Dockerfiles (nvidia/cuda 12.6.3, tinygrad 0.12.0, sshd), `mvp_entrypoint.sh` (sshd + mvp-node, held for postmortem), `local_docker_e2e.sh`, the GGUF tinygrad worker, and one-node-chat/bootstrap/vastai guarantee tests

Signed-off-by: Zachery Aaron Shores-Chmielewski <zacheryasc@gmail.com>
2026-07-01 12:44:25 +04:00

73 lines
1.7 KiB
TOML

[package]
name = "mvp-system"
version = "0.1.0"
edition = "2024"
publish = false
[features]
default = []
local-e2e = ["dep:dashboard"]
[dependencies]
datastream = { path = "../datastream" }
dashboard = { path = "../dashboard", optional = true }
serde_json = "1"
serde = { version = "1", features = ["derive"] }
swactor = { path = "../..", features = ["serde", "transport"] }
swactor-transport = { path = "../transport" }
distribution = { path = "../distribution" }
iroh-driver = { path = "../iroh-driver" }
iroh = "0.98"
tokio = { version = "1", features = ["rt-multi-thread", "macros", "process", "io-util", "sync", "time"] }
swactor-vastai = { path = "../../tools/vastai" }
parking_lot = "0.12"
[target.'cfg(target_os = "linux")'.dependencies]
libc = "0.2"
[[bin]]
name = "mvp-node"
path = "src/bin/mvp_node.rs"
[[bin]]
name = "mvp-orch-one-node"
path = "src/bin/mvp_orch_one_node.rs"
[[bin]]
name = "mvp-chat"
path = "src/bin/mvp_chat.rs"
[[bin]]
name = "mvp-one-node-chat"
path = "src/bin/mvp_one_node_chat.rs"
required-features = ["local-e2e"]
[[bin]]
name = "mvp-local-e2e"
path = "src/bin/local_e2e.rs"
required-features = ["local-e2e"]
[[bin]]
name = "mvp-local-e2e-cluster"
path = "src/bin/local_e2e_cluster.rs"
required-features = ["local-e2e"]
[[bin]]
name = "mvp-dumb-worker"
path = "src/bin/dumb_worker.rs"
required-features = ["local-e2e"]
[[test]]
name = "local_unmocked_mvp_e2e"
path = "tests/local_unmocked_mvp_e2e.rs"
required-features = ["local-e2e"]
[[test]]
name = "gpu_worker_node_e2e"
path = "tests/gpu_worker_node_e2e.rs"
required-features = ["local-e2e"]
[[test]]
name = "local-e2e-cluster"
path = "tests/local_e2e_cluster.rs"
required-features = ["local-e2e"]