Compare commits

9 Commits
m2 ... main

Author SHA1 Message Date
984a331a21 fix: increase disk space to 128GB and update OLLAMA_MODEL and MIN_VRAM_MB for improved performance 2026-05-14 11:22:55 -04:00
246c2ce102 fix: update OLLAMA_MODEL to use the latest version for improved performance 2026-05-13 22:54:42 -04:00
8e6ade213f feat: add function to fetch instance daemon logs and integrate log retrieval during instance readiness check 2026-05-13 22:35:32 -04:00
95abb4c4f5 fix: increase disk space to 64GB and handle errors during tmux session setup 2026-05-13 10:26:41 -04:00
2aef6a19c1 feat: prompt user for confirmation before destroying instance 2026-05-13 09:57:58 -04:00
f58c61d2aa fix(search_params): ensure disk_space criteria is included in search parameters 2026-05-13 09:55:08 -04:00
f641ab955f update to use latest vast api and qwen3-next on 24GB gpu 2026-05-13 09:47:02 -04:00
8bba9ef2a6 feat: pause for user input before destroying instance
Update the cleanup process in vast_start.py to require the user to press ENTER before destroying the instance. This change prevents immediate automatic termination after the subprocess ends, giving the user control over when to tear down the environment.
2026-03-23 13:09:16 -04:00
8a3d3449bf refactor(vast_start): improve structure, typing, and error handling
- Update shebang to use `#!/usr/bin/env python3`.
- Add module docstrings and type hinting across all functions.
- Move configuration variables into clearly defined uppercase constants.
- Improve error handling for file loading and API requests.
- Restructure instance creation and SSH management logic into modular, readable functions.

These changes make the script more robust, easier to maintain, and self-documenting.
2026-03-23 13:03:59 -04:00
4 changed files with 288 additions and 155 deletions

1
20 Normal file
View File

@@ -0,0 +1 @@
Error: Unknown operator. Did you forget to quote your query? ''

View File

@@ -25,7 +25,7 @@
"jupyter_dir": null,
"python_utf8": null,
"lang_utf8": null,
"disk": 32,
"disk": 64,
"last_known_min_bid": 0.6666666666666666,
"min_duration": 604800
}

View File

@@ -7,7 +7,8 @@
"cpu_ram": {"gt":23},
"gpu_ram": {"gt": 50000},
"reliability": {"gt": ".99"},
"direct": {"eq": true},
"inet_down": {"gt": 500},
"disk_space": {"gt": 200},
"order": [
[
"price",

View File

@@ -1,177 +1,308 @@
#!/bin/env python3
import requests
import time
import paramiko
#!/usr/bin/env python3
"""
Vast.ai Instance Manager
This script automates the process of finding the cheapest available GPU on Vast.ai
that meets specific VRAM requirements, creating an instance, and setting up Ollama.
"""
import json
import os
import subprocess
model="qwen3-coder-next:q8_0"
vram=80000
max_bid=2.5
space=int(vram/700+20)
import time
from typing import Any, Dict, List
def load_api_key():
path = os.path.join(
os.environ["HOME"],
".config",
"vastai",
"vast_api_key"
)
import requests
with open(path, "r") as f:
return f.read().strip()
# -----------------------------------------------------------------------------
# Configuration
# -----------------------------------------------------------------------------
OLLAMA_MODEL = "deepseek-v4-flash:cloud"
MIN_VRAM_MB = 30000
MAX_BID_HOURLY = 1.5
DISK_SPACE_GB = int(MIN_VRAM_MB * 2 / 700 + 20)
def file_obj(path):
with open(path, "r") as f:
return json.load(f)
VAST_API_BASE = "https://console.vast.ai/api/v0"
# -----------------------------------------------------------------------------
# Utility Functions
# -----------------------------------------------------------------------------
def load_api_key() -> str:
"""Loads the Vast.ai API key from the user's config directory."""
path = os.path.expanduser("~/.config/vastai/vast_api_key")
try:
with open(path, "r") as f:
return f.read().strip()
except FileNotFoundError:
raise Exception(f"API key not found at {path}. Please configure vastai CLI.")
def load_json_file(path: str) -> Dict[str, Any]:
"""Loads and parses a JSON file."""
try:
with open(path, "r") as f:
return json.load(f)
except FileNotFoundError:
raise Exception(f"Required configuration file '{path}' not found.")
# Load globals
API_KEY = load_api_key()
SEARCH_PARAMS = load_json_file("search_params.json")
CREATE_PAYLOAD = load_json_file("create_payload.json")
BASE = "https://console.vast.ai/api/v0"
params = file_obj("search_params.json")
create_payload = file_obj("create_payload.json")
def vast_request(method, path, params=None, json_data=None):
url = f"{BASE}/{path}"
print(f"Request: {method} {url} params={params} json={json_data}")
def vast_request(method: str, endpoint: str, params: Dict = None, json_data: Dict = None) -> Dict[str, Any]:
"""Sends an authenticated request to the Vast.ai API."""
url = f"{VAST_API_BASE}/{endpoint}"
headers = {"Authorization": f"Bearer {API_KEY}"}
r = requests.request(method, url, headers=headers, params=params, json=json_data)
print(f"Response: {r.status_code} {r.text}")
r.raise_for_status()
return r.json()
#print(f"[{method}] {url}")
response = requests.request(method, url, headers=headers, params=params, json=json_data)
if not response.ok:
print(f"API Error: {response.status_code} - {response.text}")
response.raise_for_status()
# Some endpoints like DELETE might return empty bodies
return response.json() if response.text else {}
def get_cheapest_gpus():
params["gpu_ram"]["gt"] = vram
r = vast_request("GET", "bundles/", params=params)
offers = r["offers"]
good_offers = []
# -----------------------------------------------------------------------------
# Core Operations
# -----------------------------------------------------------------------------
def get_cheapest_gpus() -> List[Dict[str, Any]]:
"""Searches for verified GPU offers matching our VRAM and price constraints."""
params = SEARCH_PARAMS.copy()
if "gpu_ram" not in params:
params["gpu_ram"] = {}
params["gpu_ram"]["gt"] = MIN_VRAM_MB
response = vast_request("POST", "bundles/", json_data=params)
offers = response.get("offers", [])
if not offers:
raise Exception("No GPU offers found")
for offer in offers:
if offer["gpu_ram"] > vram \
and offer["dph_total"] < max_bid \
and offer["verification"] == "verified":
good_offers.append(offer)
if len(good_offers) == 0:
raise Exception("No matching GPU offers found")
return good_offers
raise Exception("No GPU offers returned from Vast.ai.")
def create_instance(offer_id):
#create_payload['id'] = offer_id
create_payload['env']['OLLAMA_MODEL']=model
create_payload['disk']=space
r = vast_request("PUT", f"asks/{offer_id}/", json_data=create_payload)
return r["new_contract"]
def destry_instance(instance_id):
r = requests.delete(
f"{BASE}/instances/{instance_id}/",
headers={"Authorization": f"Bearer {API_KEY}"}
)
def get_instance(instance_id):
r = requests.get(
f"{BASE}/instances/{instance_id}/",
headers={"Authorization": f"Bearer {API_KEY}"}
)
data = r.json()
return data["instances"]
def wait_until_ready(instance_id):
while True:
data = get_instance(instance_id)
for tag in [
"actual_status",
"ssh_host",
"ssh_port",
"ssh_pwd",
"verification",
"uptime_mins",
"cur_state",
"direct_port_count",
"gpu_arch",
"gpu_total_ram",
"gpu_util",
"intended_status",
"next_state",
"gpu_lanes",
"gpu_name",
"public_ipaddr",
"ports",
]:
try:
if tag in data:
print(f"{tag}: {data[tag]}")
except:
print(f"{tag}: Error")
print("Instance status:", data["actual_status"])
#print("Instance data:", data)
if data["actual_status"] == "running":
return data
print("Waiting for instance...\n\n")
time.sleep(15)
def ssh_and_setup(ip, port, user, password):
commands = [
f"ollama pull {model}",
"touch ~/.no_auto_tmux",
#"ollama rm qwen3.5:35b",
#"ollama pull lama3.1:70b",
valid_offers = [
offer for offer in offers
if offer.get("gpu_ram", 0) > MIN_VRAM_MB
and offer.get("dph_total", float('inf')) < MAX_BID_HOURLY
and offer.get("verification") == "verified"
]
cmd2=["ssh","-t","-o","StrictHostKeyChecking=accept-new","-p",str(port),f"{user}@{ip}", "touch","~/.no_auto_tmux"]
subprocess.run(cmd2)
cmd2=["ssh","-t","-p",str(port),f"{user}@{ip}","ollama","pull",model]
print(' '.join(cmd2))
subprocess.run(cmd2)
if not valid_offers:
raise Exception("No verified GPU offers found matching the VRAM and price constraints.")
# Ensure they are sorted by price (cheapest first)
valid_offers.sort(key=lambda x: x.get("dph_total", float('inf')))
return valid_offers
def create_instance(offer_id: int) -> int:
"""Creates a new instance from a specific offer ID."""
payload = CREATE_PAYLOAD.copy()
if "env" not in payload:
payload["env"] = {}
payload["env"]["OLLAMA_MODEL"] = OLLAMA_MODEL
payload["disk"] = DISK_SPACE_GB
response = vast_request("PUT", f"asks/{offer_id}/", json_data=payload)
if "new_contract" not in response:
raise Exception(f"Failed to create instance. API Response: {response}")
return response["new_contract"]
def destroy_instance(instance_id: int) -> None:
"""Destroys a running instance."""
print(f"Destroying instance {instance_id}...")
vast_request("DELETE", f"instances/{instance_id}/")
print(f"Instance {instance_id} destroyed.")
def main():
print("Searching cheapest 39GB+ GPU...")
offers = get_cheapest_gpus()
instance_id=0
for offer in offers:
offer_id = offer["id"]
print("Offer:", offer_id, "GPU:", offer["gpu_name"], "Price:", offer["dph_total"])
print("Creating instance...")
try:
for k in offer:
v=offer[k]
print(f"{k}: {v}")
instance_id = create_instance(offer_id)
def get_instance_details(instance_id: int) -> Dict[str, Any]:
"""Fetches current details and status of an instance."""
response = vast_request("GET", f"instances/{instance_id}/")
instances = response.get("instances", [])
if not instances:
raise Exception(f"Instance {instance_id} not found.")
return instances
def get_instance_demon_logs(instance_id: int) -> str:
"""Fetches the latest logs for an instance."""
js_data={'daemon_logs': 'true'}
log_meta = vast_request("PUT", f"instances/request_logs/{instance_id}/", json_data=js_data)
temp_download_url = log_meta.get("temp_download_url")
if not temp_download_url:
print(json.dumps(log_meta, indent=2))
raise Exception(f"Failed to get logs for instance {instance_id}. API Response: {log_meta}")
log_response=None
for _ in range(15):
log_response = requests.get( temp_download_url,timeout=30, headers={"cache-control": "no-cache"})
if log_response.ok:
break
except:
print(f"failed to claim offer {offer_id}")
data=[]
if instance_id > 0 :
print("Instance:", instance_id)
time.sleep(1)
if not log_response.ok:
raise Exception(f"Failed to download logs from {temp_download_url}. Status: {log_response.status_code}")
return log_response.text
def get_instances() -> List[Dict[str, Any]]:
"""Fetches a list of all instances for the user."""
response = vast_request("GET", "instances/")
return response.get("instances", [])
def wait_until_ready(instance_id: int, poll_interval: int = 15) -> Dict[str, Any]:
"""Polls the instance status until it is running and ready."""
print(f"Waiting for instance {instance_id} to become ready...")
# Tags we care to monitor during startup
status_tags = [
"actual_status", "intended_status", "cur_state", "next_state",
"gpu_name", "gpu_util", "public_ipaddr", "ssh_port", "uptime_mins"
]
while True:
data = get_instance_details(instance_id)
status = data.get("actual_status", "unknown")
# Print a summarized status line
info = []
for tag in status_tags:
if tag in data and data[tag] is not None:
info.append(f"{tag}={data[tag]}")
#print(f"Status: {status} | {' | '.join(info)}")
if status == "running":
# Extra check to ensure SSH port is populated
if data.get("ssh_port") and data.get("ssh_host"):
print("Instance is running and network is configured.")
time.sleep(5) # Brief pause to ensure services are up
return data
else:
print("Instance running, but waiting for SSH details to populate...")
#print("Instance not ready yet. Fetching logs for insights...")
try:
logs = get_instance_demon_logs(instance_id)
print(chr(27) + "[2J")
print(f"--- Logs for Instance {instance_id} ---")
print(f"Status: {status} | {' | '.join(info)}")
#print("instanxce_details:"+json.dumps(data))
#print(f"Current Status: {status}")
print(f"Status MSG: {data['status_msg']}")
print(f"instance_id: {instance_id}")
print(f"Date: {time.strftime('%Y-%m-%d %H:%M:%S')}\n\n")
print("Retrieved logs:")
print('\n'.join(logs.splitlines()[-30:])) # Print last 10 lines for brevity
except Exception as e:
print(f"Error fetching logs: {e}")
time.sleep(poll_interval)
def ssh_and_setup(ip: str, port: int, user: str = "root") -> None:
"""Connects to the instance via SSH and sets up the environment."""
print(f"Setting up instance via SSH at {user}@{ip}:{port}")
ssh_base_cmd = [
"ssh", "-t",
"-o", "StrictHostKeyChecking=accept-new",
"-p", str(port),
f"{user}@{ip}"
]
# Disable auto-tmux
subprocess.run(ssh_base_cmd + ["touch", "~/.no_auto_tmux"], check=True)
try:
subprocess.run(ssh_base_cmd + ["tmux","new","-d","-s","ollama", "ollama serve"], check=True)
except subprocess.CalledProcessError as e:
print(f"Error occurred while setting up tmux session: {e}")
# Pull the required Ollama model
pull_cmd = ssh_base_cmd + ["ollama", "pull", OLLAMA_MODEL]
print(f"Executing: {' '.join(pull_cmd)}")
subprocess.run(pull_cmd, check=True)
# -----------------------------------------------------------------------------
# Main Execution
# -----------------------------------------------------------------------------
def main():
instances = get_instances()
instance_id = 0
if instances:
print("Existing instances found:")
for inst in instances:
print(f" - ID: {inst['id']}, Status: {inst['actual_status']}, GPU: {inst.get('gpu_name', 'N/A')}")
instance_id = inst["id"]
print("Searching for the cheapest suitable GPU...")
try:
offers = get_cheapest_gpus()
except Exception as e:
print(f"Error finding GPUs: {e}")
return
if instance_id == 0:
for offer in offers:
offer_id = offer["id"]
gpu_name = offer.get("gpu_name", "Unknown GPU")
price = offer.get("dph_total", 0.0)
print(f"\nAttempting to claim Offer {offer_id} (GPU: {gpu_name}, Price: ${price:.3f}/hr)")
try:
instance_id = create_instance(offer_id)
print(f"Successfully claimed offer. Instance ID: {instance_id}")
break
except Exception as e:
print(f"Failed to claim offer {offer_id}: {e}")
if not instance_id or instance_id == 0:
print("Failed to claim any of the available offers.")
return
try:
# Wait for initialization
data = wait_until_ready(instance_id)
ip = data["ssh_host"]
port = data["ssh_port"]
user = "root"
#password = data["ssh_pwd"]
# Extract connection details
ssh_host = data.get("ssh_host")
ssh_port = data.get("ssh_port")
public_ip = data.get("public_ipaddr")
ssh_host=public_ip if public_ip else ssh_host
ports_config = data.get("ports", {})
pub_ssh_port = ports_config.get("22/tcp", [{}])[0].get("HostPort", ssh_port)
pub_ollama_port = ports_config.get("11434/tcp", [{}])[0].get("HostPort", "Unknown")
print("Instance ready:", ip)
print(f"\nInstance is fully ready at {ssh_host}:{ssh_port}")
# Setup SSH and Ollama
ssh_and_setup(public_ip, pub_ssh_port, "root")
print("\n--- Setup Complete ---")
print(f"SSH Command: ssh root@{public_ip} -p {pub_ssh_port}")
print(f"Ollama Endpoint: http://{public_ip}:{pub_ollama_port} (Model: {OLLAMA_MODEL})")
# Run interactive shell or prompt
print(f"\nStarting interactive Ollama session for model: {OLLAMA_MODEL}")
run_cmd = [
"ssh", "-t",
"-o", "StrictHostKeyChecking=accept-new",
"-p", str(pub_ssh_port),
f"root@{public_ip}",
"ollama", "run", OLLAMA_MODEL
]
print(f"Executing: {' '.join(run_cmd)}")
subprocess.run(run_cmd)
pub_ip=data["public_ipaddr"]
pub_port=data['ports']['22/tcp'][0]['HostPort']
ssh_and_setup(ip, port, user, "")
print(f"ssh root@{pub_ip} -p {pub_port}")
ollama_port=data['ports']['11434/tcp'][0]['HostPort']
print(f"ollama up at http://{pub_ip}:{ollama_port} with mode {model}")
cmd2=["ssh","-t","-p",str(pub_port),f"{user}@{pub_ip}","ollama","run",model]
print(' '.join(cmd2))
subprocess.run(cmd2)
except KeyboardInterrupt:
print("\nOperation cancelled by user.")
except Exception as e:
print(f"\nAn error occurred during instance setup: {e}")
finally:
if instance_id:
try:
should_destroy=input("\nENTER D to destroy the instance and clean up...")
except (EOFError, KeyboardInterrupt):
pass
if should_destroy.strip().lower() == 'd':
print("\nCleaning up...")
destroy_instance(instance_id)
if __name__ == "__main__":
main()
main()