File: //proc/self/root/tmp/ezos-check.out
{
"updated": "2026-05-31",
"services": [
{
"name": "Managed Local AI",
"url": "/managed-local-ai/",
"summary": "Managed Ollama-based local AI on customer-owned or customer-rented GPU infrastructure with Open WebUI as the default interface.",
"cta": "Order Managed Local AI",
"href": "https://support.ezoshosting.com/cart/managed-local-ai/",
"bullets": [
"Ollama standard",
"Open WebUI default",
"RTX 4000 Ada class 20 GB model-fit shortlist",
"Benchmark report before throughput promises",
"No third-party AI API required by default",
"Private RAG with source-answer smoke tests",
"Local document and transcription workflow benchmarks",
"OpenAI-compatible local API bridge benchmark",
"Open WebUI RBAC and SSO/OIDC access scope"
],
"plans": [
{
"name": "BYO Server Management",
"from_usd_monthly": 299.18,
"fit": "Customer-owned or customer-rented GPU server with managed Ollama/Open WebUI operations."
},
{
"name": "Local AI Managed",
"from_usd_monthly": 699.42,
"fit": "Managed private local model hosting with benchmark-first setup."
},
{
"name": "Team RAG",
"from_usd_monthly": 999.6,
"fit": "Document-assisted local AI and team knowledge workflows."
},
{
"name": "Business Secure",
"from_usd_monthly": 1499.9,
"fit": "Controlled production rollout with security hardening, audit preparation, and support scope."
}
],
"runtime_policy": {
"local_ai": "benchmark-first",
"gpu_class": "RTX 4000 Ada class 20 GB systems are positioned for small-to-medium local models.",
"readiness_gate": "Do not promise live local inference until GPU driver visibility, Ollama service health, and a model smoke test pass."
},
"current_model_shortlist": [
{
"use_case": "team_rag_embeddings_and_reranking",
"candidates": [
"Qwen3-Embedding 0.6B",
"Qwen3-Embedding 4B",
"Qwen3-Embedding 8B",
"Qwen3-Reranker 0.6B",
"Qwen3-Reranker 4B",
"Qwen3-Reranker 8B"
],
"positioning": "Multilingual retrieval, code search, source ranking, and private knowledge bases with measured storage, latency, and cited-answer behavior."
},
{
"use_case": "document_intake_vision",
"candidates": [
"Qwen3-VL 8B",
"Qwen2.5-VL 7B fallback",
"Docling/OCR baseline"
],
"positioning": "Private scanned PDF, screenshot, invoice, form, table, reading-order, and field-level extraction benchmark before workflow automation."
},
{
"use_case": "code_assistant_benchmark",
"candidates": [
"Qwen3-Coder 30B",
"Qwen3 14B fallback"
],
"positioning": "Benchmark-only private repository assistant path on RTX 4000 Ada class 20 GB systems; context, concurrency, latency, and access controls decide fit."
},
{
"use_case": "assistant_and_support",
"candidates": [
"Qwen3 8B/14B",
"Gemma 3 12B"
],
"positioning": "First-pass candidates for support assistants, summaries, drafting, and RAG answers before larger or specialized model trials."
}
],
"market_positioning": "Benchmark current Qwen3 Embedding/Reranker, Qwen3-VL, Qwen3-Coder, Gemma, Docling/OCR, OpenAI-compatible local API bridge, and Open WebUI access-control candidates on RTX 4000 Ada class 20 GB systems before promising throughput.",
"offer_tracks": [
{
"name": "Private RAG with sources",
"href": "https://support.ezoshosting.com/cart/managed-local-ai/Team-RAG/&step=0",
"fit": "Internal documents, support archives, policies, and project knowledge with cited answers.",
"stack_candidates": [
"Open WebUI",
"Qwen or Gemma chat candidate",
"EmbeddingGemma or Qwen embedding trial"
],
"smoke_test": "Representative corpus ingest, fixed benchmark questions, cited-answer requirement, VRAM, latency, and miss-behavior log."
},
{
"name": "Local PDF and image extraction",
"href": "https://support.ezoshosting.com/cart/managed-local-ai/Business-Secure/&step=0",
"fit": "Invoices, forms, screenshots, and operational documents that need local extraction support.",
"stack_candidates": [
"Vision-language model benchmark",
"OCR/PDF fallback",
"Field-level validation notes"
],
"smoke_test": "Real sample pages against expected fields, false-positive review, unsupported layout notes, throughput, and VRAM peak."
},
{
"name": "Local transcription and meeting notes",
"href": "https://support.ezoshosting.com/cart/managed-local-ai/Business-Secure/&step=0",
"fit": "Interviews, internal meetings, and support recordings with private audio handling.",
"stack_candidates": [
"Whisper or faster-whisper benchmark",
"English and German sample set",
"Optional local summary pass"
],
"smoke_test": "Representative 5 to 30 minute audio files, runtime factor, language quality, segmentation limits, and GPU use."
},
{
"name": "Private AI app bridge",
"href": "https://support.ezoshosting.com/cart/managed-local-ai/Business-Secure/&step=0",
"fit": "Internal applications, scripts, agent tools, and prototypes that need a local OpenAI-compatible endpoint with controlled access.",
"stack_candidates": [
"Ollama OpenAI-compatible API",
"Open WebUI RBAC/SSO scope",
"Optional vLLM quantized serving trial"
],
"smoke_test": "Run representative app calls, confirm access boundaries, log latency and VRAM, and document fallback behavior before production traffic."
}
]
},
{
"name": "AI Apps",
"url": "/ai-apps/",
"summary": "Private knowledge, team chat, workflow automation, and creative GPU app stacks managed around open-source tools.",
"cta": "Compare AI Apps",
"href": "/ai-apps/",
"bullets": [
"AnythingLLM or LibreChat",
"Flowise and n8n options",
"ComfyUI for creative GPU workflows",
"vLLM only as optional advanced layer",
"Team RAG benchmark path",
"Qwen3-Embedding retrieval candidates",
"Permissions and update-window scoping",
"Code assistant benchmark path",
"Qwen3-Coder 30B trial with fallbacks",
"OpenAI-compatible local endpoint benchmark",
"Open WebUI RBAC and SSO/OIDC scoping"
],
"team_rag_fastpath": {
"href": "https://support.ezoshosting.com/cart/managed-local-ai/Team-RAG/&step=0",
"from_usd_monthly": 999.6,
"benchmark_candidates": [
"Qwen3-Embedding 0.6B",
"Qwen3-Embedding 4B",
"Qwen3-Embedding 8B"
],
"positioning": "Private knowledge benchmark and deployment path; no live inference promise until GPU, Ollama, and target-model smoke tests pass."
},
"code_assistant_fastpath": {
"href": "https://support.ezoshosting.com/cart/managed-local-ai/Business-Secure/&step=0",
"benchmark_candidates": [
"Qwen3-Coder 30B",
"Qwen3 14B fallback"
],
"positioning": "Private repository assistant benchmark; no live production claim until GPU, Ollama, and target-model smoke tests pass."
}
},
{
"name": "GPU Infrastructure",
"url": "/gpu-infrastructure/",
"summary": "Right-sized GPU servers for local inference, image workflows, and private automation stacks.",
"cta": "Plan GPU Stack",
"href": "/gpu-infrastructure/",
"bullets": [
"Dedicated setup",
"Storage and backup planning",
"Monitoring and maintenance",
"RTX 4000 Ada class fit guidance",
"Benchmark before performance promises"
]
},
{
"name": "Open Source Hosting",
"url": "/open-source-hosting/",
"summary": "Managed open-source hosting with CyberPanel, domains, SSL, DNS, and human support.",
"cta": "View Hosting Options",
"href": "https://support.ezoshosting.com/cart/open-source-hosting/",
"bullets": [
"CyberPanel control panel",
"WordPress, Nextcloud, Matomo and more",
"No cPanel or CentOS claims",
"Built for long-term maintenance",
"Starter, Pro, Business, and Managed App paths",
"Checkout remains final for term, options, renewal, taxes, and total"
],
"plans": [
{
"name": "Open Source Starter",
"from_usd_monthly": 12.91,
"fit": "One small site, project, or association with maintainable open-source hosting."
},
{
"name": "Open Source Pro",
"from_usd_monthly": 29.91,
"fit": "Active WordPress, Nextcloud, Matomo, or similar open-source applications."
},
{
"name": "Open Source Business",
"from_usd_monthly": 69.04,
"fit": "Business sites needing predictable support, restore help, and operational review."
},
{
"name": "Managed App Hosting",
"from_usd_monthly": 99.06,
"fit": "Managed application layer with updates, backups, monitoring setup, and maintenance planning."
}
],
"pricing_note": "Public starting prices observed in the HostBill cart on 2026-05-31; checkout remains authoritative for term, options, renewal, taxes, and final total."
},
{
"name": "Domains",
"url": "/domains/",
"summary": "Domain registration, renewal, transfer guidance, DNS support, and secure live availability and price checks for project domains.",
"cta": "Check live availability and price",
"href": "https://support.ezoshosting.com/checkdomain/domains/",
"bullets": [
"Secure checker confirms final USD price",
"Dated examples support quick comparison",
"Transfers reviewed by registry rules",
"DNS basics included"
],
"pricing_note": "Popular TLD examples were observed in the public HostBill checker on 2026-05-31; the secure checker and checkout remain authoritative for availability, term, taxes, and final price."
},
{
"name": "Support",
"url": "/support/",
"summary": "Human support, knowledgebase, client area, and ticket flow for managed hosting and AI stacks.",
"cta": "Get Support",
"href": "/support/",
"bullets": [
"Knowledgebase entry point",
"Tickets for clients",
"Managed setup consultation",
"Real operational help"
]
},
{
"name": "Affiliates",
"url": "/affiliates/",
"summary": "A partner program for people who recommend hosting and managed local AI they actually trust.",
"cta": "Join Affiliates",
"href": "https://support.ezoshosting.com/affiliates/",
"bullets": [
"Trackable partner flow",
"No fake claims",
"Eligible paid orders only",
"Transparent program language"
]
}
],
"runtime_policy": {
"local_ai": "benchmark-first",
"gpu_class": "RTX 4000 Ada class 20 GB systems are positioned for small-to-medium local models.",
"readiness_gate": "Do not promise live local inference until GPU driver visibility, Ollama service health, and a model smoke test pass."
},
"market_watch": {
"last_checked": "2026-05-31",
"sources": [
"Ollama Qwen3 library",
"Ollama Gemma 3 library",
"Ollama Qwen3 Embedding library",
"Open WebUI Ollama documentation",
"NVIDIA RTX 4000 Ada specifications",
"Ollama Qwen3-Coder library",
"Ollama EmbeddingGemma library",
"vLLM NVIDIA CUDA installation documentation",
"Hugging Face Qwen3-8B",
"Hugging Face Qwen3-Embedding-8B",
"Hugging Face Qwen2.5-VL-7B-Instruct",
"Hugging Face Whisper large-v3",
"SYSTRAN faster-whisper",
"Qwen3 Embedding official blog",
"Hugging Face Qwen3-VL-8B-Instruct",
"Docling official project"
],
"operational_note": "GPU runtime is gated until driver visibility, Ollama health, and target-model smoke tests pass on the actual host.",
"offer_track_note": "Next commercial tracks are benchmarked setup packages: source-backed RAG with reranking, local document/image extraction with Docling/OCR baselines, local transcription, and private code assistant trials. Claims remain gated by runtime and smoke-test evidence."
}
}