Initial SQL-only 1C adapter baseline

This commit is contained in:
2026-07-22 03:03:47 +03:00
commit e2503b77e7
545 changed files with 184711 additions and 0 deletions
+19
View File
@@ -0,0 +1,19 @@
# Model Registry
Реестр хранит только описания моделей, адаптеров и их версий.
Большие файлы моделей не должны попадать в git. Для каждой модели создается `model-card.yaml` в `registry/model-cards`.
Минимальные поля:
- `id`
- `type`
- `task`
- `language`
- `source`
- `license`
- `storage_path`
- `vram_required_gb`
- `status`
Шаблон находится в `registry/templates/model-card.yaml`.
+473
View File
@@ -0,0 +1,473 @@
{
"schema_version": 1,
"models": [
{
"id": "animagine-xl-4_0",
"name": "Animagine XL 4.0",
"type": "image-diffusion-model",
"status": "candidate",
"task": [
"image-generation"
],
"language": [
"en",
"multilingual"
],
"source": "huggingface",
"upstream_id": "cagliostrolab/animagine-xl-4.0",
"license": "openrail++",
"storage_path": "/models/image/animagine-xl-4.0",
"format": "diffusers",
"quantization": "fp16",
"runtime": "transformers",
"served_model_name": "animagine-xl-4",
"card_path": "registry/model-cards/animagine-xl-4_0.yaml"
},
{
"id": "devstral-small-2-24b-instruct-2512-q4_k_m",
"name": "Devstral Small 2 24B Instruct 2512 GGUF Q4_K_M",
"type": "gguf-model",
"status": "candidate",
"task": [
"1c",
"code",
"agentic-coding",
"text",
"tool-use"
],
"language": [
"ru",
"en"
],
"source": "huggingface",
"upstream_id": "bartowski/mistralai_Devstral-Small-2-24B-Instruct-2512-GGUF",
"license": "apache-2.0",
"storage_path": "/models/gguf/1c/devstral-small-2-24b-instruct-2512-q4_k_m",
"format": "gguf",
"quantization": "Q4_K_M",
"runtime": "llama.cpp",
"served_model_name": "devstral-1c-q4",
"card_path": "registry/model-cards/devstral-small-2-24b-instruct-2512-q4_k_m.yaml"
},
{
"id": "illustriousxl",
"name": "Illustrious XL",
"type": "image-diffusion-model",
"status": "candidate",
"task": [
"image-generation"
],
"language": [
"en",
"multilingual"
],
"source": "huggingface",
"upstream_id": "glides/illustriousxl",
"license": "mit",
"storage_path": "/models/image/illustriousxl",
"format": "diffusers",
"quantization": "fp16",
"runtime": "transformers",
"served_model_name": "illustriousxl",
"card_path": "registry/model-cards/illustriousxl.yaml"
},
{
"id": "lmt-60-4b",
"name": "LMT-60 4B",
"type": "translation-model",
"status": "candidate",
"task": [
"translation"
],
"language": [
"ru",
"en",
"multilingual"
],
"source": "huggingface",
"upstream_id": "NiuTrans/LMT-60-4B",
"license": "apache-2.0",
"storage_path": "/models/translation/lmt-60-4b",
"format": "safetensors",
"quantization": "none",
"runtime": "transformers",
"served_model_name": "lmt-60-4b",
"card_path": "registry/model-cards/lmt-60-4b.yaml"
},
{
"id": "qwen-image-edit",
"name": "Qwen Image Edit",
"type": "image-diffusion-model",
"status": "candidate",
"task": [
"image-editing",
"inpainting",
"image-generation"
],
"language": [
"ru",
"en",
"multilingual"
],
"source": "huggingface",
"upstream_id": "Qwen/Qwen-Image-Edit",
"license": "apache-2.0",
"storage_path": "/models/image/qwen-image-edit",
"format": "diffusers",
"quantization": "bf16",
"runtime": "transformers",
"served_model_name": "qwen-image-edit",
"card_path": "registry/model-cards/qwen-image-edit.yaml"
},
{
"id": "qwen2_5-vl-7b-instruct",
"name": "Qwen2.5-VL 7B Instruct",
"type": "vision-language-model",
"status": "candidate",
"task": [
"video",
"image-understanding",
"document-understanding",
"visual-question-answering"
],
"language": [
"ru",
"en",
"multilingual"
],
"source": "huggingface",
"upstream_id": "Qwen/Qwen2.5-VL-7B-Instruct",
"license": "apache-2.0",
"storage_path": "/models/video/qwen2.5-vl-7b-instruct",
"format": "safetensors",
"quantization": "none",
"runtime": "transformers",
"served_model_name": "qwen2.5-vl-7b-instruct",
"card_path": "registry/model-cards/qwen2_5-vl-7b-instruct.yaml"
},
{
"id": "qwen3-14b-instruct-q6_k",
"name": "Qwen3 14B Instruct GGUF Q6_K",
"type": "gguf-model",
"status": "candidate",
"task": [
"text",
"chat",
"summarization",
"code",
"tool-use",
"1c-rag"
],
"language": [
"ru",
"en",
"multilingual"
],
"source": "huggingface",
"upstream_id": "Qwen/Qwen3-14B-GGUF",
"license": "apache-2.0",
"storage_path": "/models/gguf/text/qwen3-14b-instruct-q6_k",
"format": "gguf",
"quantization": "Q6_K",
"runtime": "llama.cpp",
"served_model_name": "qwen3-14b-q6",
"card_path": "registry/model-cards/qwen3-14b-instruct-q6_k.yaml"
},
{
"id": "qwen3-4b-1c-lora-v1",
"name": "Qwen3 4B 1C LoRA v1",
"type": "lora-adapter",
"status": "draft",
"task": [
"1c",
"bsl-code",
"metadata-safety",
"1c-query",
"explanation"
],
"language": [
"ru"
],
"source": "local-training",
"upstream_id": null,
"license": "internal",
"storage_path": "/models/adapters/1c/qwen3-4b-1c-lora-v1",
"format": "safetensors",
"quantization": null,
"runtime": "vllm",
"served_model_name": "qwen3-4b-1c",
"card_path": "registry/model-cards/qwen3-4b-1c-lora-v1.yaml"
},
{
"id": "qwen3-4b-instruct-2507",
"name": "Qwen3 4B Instruct 2507",
"type": "base-model",
"status": "staging",
"task": [
"text",
"chat",
"summarization",
"code",
"tool-use",
"1c-rag"
],
"language": [
"ru",
"en"
],
"source": "huggingface",
"upstream_id": "Qwen/Qwen3-4B-Instruct-2507",
"license": "apache-2.0",
"storage_path": "/models/base/qwen3-4b-instruct-2507",
"format": "safetensors",
"quantization": "none",
"runtime": "vllm",
"served_model_name": "qwen3-4b-instruct",
"card_path": "registry/model-cards/qwen3-4b-instruct-2507.yaml"
},
{
"id": "qwen3-coder-30b-a3b-1c-lora-v1",
"name": "Qwen3 Coder 30B A3B 1C LoRA v1",
"type": "lora-adapter",
"status": "draft",
"task": [
"1c",
"bsl-code",
"metadata-safety",
"metadata-write-verified",
"metadata-write-learning-plan",
"1c-query",
"explanation"
],
"language": [
"ru"
],
"source": "local-training",
"upstream_id": null,
"license": "internal",
"storage_path": "/models/adapters/1c/qwen3-coder-30b-a3b-1c-lora-v1",
"format": "safetensors",
"quantization": null,
"runtime": "llama.cpp",
"served_model_name": "qwen3-coder-1c-q6",
"card_path": "registry/model-cards/qwen3-coder-30b-a3b-1c-lora-v1.yaml"
},
{
"id": "qwen3-coder-30b-a3b-instruct-q4_k_m",
"name": "Qwen3 Coder 30B A3B Instruct GGUF Q4_K_M",
"type": "gguf-model",
"status": "candidate",
"task": [
"1c",
"code",
"agentic-coding",
"repository-analysis",
"tool-use"
],
"language": [
"ru",
"en"
],
"source": "huggingface",
"upstream_id": "lmstudio-community/Qwen3-Coder-30B-A3B-Instruct-GGUF",
"license": "apache-2.0",
"storage_path": "/models/gguf/1c/qwen3-coder-30b-a3b-instruct-q4_k_m",
"format": "gguf",
"quantization": "Q4_K_M",
"runtime": "llama.cpp",
"served_model_name": "qwen3-coder-1c-q4",
"card_path": "registry/model-cards/qwen3-coder-30b-a3b-instruct-q4_k_m.yaml"
},
{
"id": "qwen3-coder-30b-a3b-instruct-q6_k",
"name": "Qwen3 Coder 30B A3B Instruct GGUF Q6_K",
"type": "gguf-model",
"status": "candidate",
"task": [
"1c",
"code",
"agentic-coding",
"repository-analysis",
"tool-use"
],
"language": [
"ru",
"en"
],
"source": "huggingface",
"upstream_id": "lmstudio-community/Qwen3-Coder-30B-A3B-Instruct-GGUF",
"license": "apache-2.0",
"storage_path": "/models/gguf/1c/qwen3-coder-30b-a3b-instruct-q6_k",
"format": "gguf",
"quantization": "Q6_K",
"runtime": "llama.cpp",
"served_model_name": "qwen3-coder-1c-q6",
"card_path": "registry/model-cards/qwen3-coder-30b-a3b-instruct-q6_k.yaml"
},
{
"id": "qwen3-coder-30b-a3b-instruct",
"name": "Qwen3 Coder 30B A3B Instruct",
"type": "base-model",
"status": "staging",
"task": [
"1c",
"code",
"agentic-coding",
"repository-analysis",
"tool-use"
],
"language": [
"ru",
"en"
],
"source": "huggingface",
"upstream_id": "Qwen/Qwen3-Coder-30B-A3B-Instruct",
"license": "apache-2.0",
"storage_path": "/models/base/qwen3-coder-30b-a3b-instruct",
"format": "safetensors",
"quantization": "none",
"runtime": "training",
"served_model_name": "qwen3-coder-30b-a3b-instruct",
"card_path": "registry/model-cards/qwen3-coder-30b-a3b-instruct.yaml"
},
{
"id": "qwen3-vl-8b-instruct",
"name": "Qwen3-VL 8B Instruct",
"type": "vision-language-model",
"status": "candidate",
"task": [
"video",
"image-understanding",
"document-understanding",
"visual-question-answering"
],
"language": [
"ru",
"en",
"multilingual"
],
"source": "huggingface",
"upstream_id": "Qwen/Qwen3-VL-8B-Instruct",
"license": "apache-2.0",
"storage_path": "/models/video/qwen3-vl-8b-instruct",
"format": "safetensors",
"quantization": "none",
"runtime": "transformers",
"served_model_name": "qwen3-vl-8b-instruct",
"card_path": "registry/model-cards/qwen3-vl-8b-instruct.yaml"
},
{
"id": "realvisxl-v5_0",
"name": "RealVisXL V5.0",
"type": "image-diffusion-model",
"status": "candidate",
"task": [
"image-generation"
],
"language": [
"en",
"multilingual"
],
"source": "huggingface",
"upstream_id": "SG161222/RealVisXL_V5.0",
"license": "openrail++",
"storage_path": "/models/image/realvisxl-v5.0",
"format": "diffusers",
"quantization": "fp16",
"runtime": "transformers",
"served_model_name": "realvisxl-v5",
"card_path": "registry/model-cards/realvisxl-v5_0.yaml"
},
{
"id": "sdxl-base-1_0",
"name": "Stable Diffusion XL Base 1.0",
"type": "image-diffusion-model",
"status": "staging",
"task": [
"image-generation"
],
"language": [
"en",
"multilingual"
],
"source": "huggingface",
"upstream_id": "stabilityai/stable-diffusion-xl-base-1.0",
"license": "openrail++",
"storage_path": "/models/image/sdxl-base-1.0",
"format": "diffusers",
"quantization": "fp16",
"runtime": "transformers",
"served_model_name": "sdxl-image",
"card_path": "registry/model-cards/sdxl-base-1_0.yaml"
},
{
"id": "sdxl-inpainting-1_0",
"name": "Stable Diffusion XL Inpainting 1.0",
"type": "image-diffusion-model",
"status": "staging",
"task": [
"image-editing",
"inpainting"
],
"language": [
"en",
"multilingual"
],
"source": "huggingface",
"upstream_id": "diffusers/stable-diffusion-xl-1.0-inpainting-0.1",
"license": "openrail++",
"storage_path": "/models/image/sdxl-inpainting-1.0",
"format": "diffusers",
"quantization": "fp16",
"runtime": "transformers",
"served_model_name": "sdxl-image",
"card_path": "registry/model-cards/sdxl-inpainting-1_0.yaml"
},
{
"id": "whisper-large-v3-turbo-russian",
"name": "Whisper Large v3 Turbo Russian",
"type": "speech-model",
"status": "candidate",
"task": [
"speech-to-text",
"speech-translation"
],
"language": [
"ru"
],
"source": "huggingface",
"upstream_id": "dvislobokov/whisper-large-v3-turbo-russian",
"license": "mit",
"storage_path": "/models/audio/whisper-large-v3-turbo-russian",
"format": "safetensors",
"quantization": "none",
"runtime": "transformers",
"served_model_name": "whisper-large-v3-turbo-russian",
"card_path": "registry/model-cards/whisper-large-v3-turbo-russian.yaml"
},
{
"id": "whisper-large-v3-turbo",
"name": "Whisper Large v3 Turbo",
"type": "speech-model",
"status": "candidate",
"task": [
"speech-to-text",
"speech-translation"
],
"language": [
"ru",
"en",
"multilingual"
],
"source": "huggingface",
"upstream_id": "openai/whisper-large-v3-turbo",
"license": "mit",
"storage_path": "/models/audio/whisper-large-v3-turbo",
"format": "safetensors",
"quantization": "none",
"runtime": "transformers",
"served_model_name": "whisper-large-v3-turbo",
"card_path": "registry/model-cards/whisper-large-v3-turbo.yaml"
}
]
}
+1
View File
@@ -0,0 +1 @@
@@ -0,0 +1,29 @@
id: animagine-xl-4_0
name: Animagine XL 4.0
type: image-diffusion-model
task:
- image-generation
language:
- en
- multilingual
source: huggingface
upstream_id: cagliostrolab/animagine-xl-4.0
upstream_url: https://huggingface.co/cagliostrolab/animagine-xl-4.0
license: openrail++
status: candidate
created_at: 2026-07-04
storage_path: /models/image/animagine-xl-4.0
format: diffusers
quantization: fp16
context_length: null
vram_required_gb: 12
parameters: null
base_model: stabilityai/stable-diffusion-xl-base-1.0
adapter_for: null
datasets: []
eval_suites: []
deployment:
target: docker-gpu.cin.su
runtime: transformers
served_model_name: animagine-xl-4
notes: "Anime-focused SDXL finetune with Diffusers support on Hugging Face. Good fit for RTX 4090 and compatible with the current StableDiffusionXLPipeline-based image service."
@@ -0,0 +1,38 @@
id: devstral-small-2-24b-instruct-2512-q4_k_m
name: Devstral Small 2 24B Instruct 2512 GGUF Q4_K_M
type: gguf-model
task:
- 1c
- code
- agentic-coding
- text
- tool-use
language:
- ru
- en
source: huggingface
upstream_id: bartowski/mistralai_Devstral-Small-2-24B-Instruct-2512-GGUF
upstream_url: https://huggingface.co/bartowski/mistralai_Devstral-Small-2-24B-Instruct-2512-GGUF
license: apache-2.0
status: candidate
created_at: 2026-06-18
storage_path: /models/gguf/1c/devstral-small-2-24b-instruct-2512-q4_k_m
filename: mistralai_Devstral-Small-2-24B-Instruct-2512-Q4_K_M.gguf
file_size_bytes: 14334438272
format: gguf
quantization: Q4_K_M
context_length: null
deployment_context_length: 32768
vram_required_gb: null
parameters: 24B
base_model: mistralai/Devstral-Small-2-24B-Instruct-2512
adapter_for: null
datasets: []
eval_suites:
- plugins/1c/evals/smoke.yaml
deployment:
target: docker-gpu.cin.su
runtime: llama.cpp
served_model_name: devstral-1c-q4
compose: core/deploy/docker-gpu/llama-cpp/compose.yaml
notes: "GGUF Q4_K_M quant for 1C/code experiments. Loads on docker-gpu.cin.su with llama.cpp CUDA and uses GPU, but current llama.cpp server returns empty decoded content for chat/completion despite token generation. Runtime is experimental until a compatible template/runtime/quant is found."
+29
View File
@@ -0,0 +1,29 @@
id: illustriousxl
name: Illustrious XL
type: image-diffusion-model
task:
- image-generation
language:
- en
- multilingual
source: huggingface
upstream_id: glides/illustriousxl
upstream_url: https://huggingface.co/glides/illustriousxl
license: mit
status: candidate
created_at: 2026-07-04
storage_path: /models/image/illustriousxl
format: diffusers
quantization: fp16
context_length: null
vram_required_gb: 12
parameters: null
base_model: stabilityai/stable-diffusion-xl-base-1.0
adapter_for: null
datasets: []
eval_suites: []
deployment:
target: docker-gpu.cin.su
runtime: transformers
served_model_name: illustriousxl
notes: "Stylized illustration / anime-oriented SDXL model with explicit StableDiffusionXLPipeline usage on Hugging Face. Useful as a third visual style alongside photorealism and pure anime."
+30
View File
@@ -0,0 +1,30 @@
id: lmt-60-4b
name: LMT-60 4B
type: translation-model
task:
- translation
language:
- ru
- en
- multilingual
source: huggingface
upstream_id: NiuTrans/LMT-60-4B
upstream_url: https://huggingface.co/NiuTrans/LMT-60-4B
license: apache-2.0
status: candidate
created_at: 2026-06-19
storage_path: /models/translation/lmt-60-4b
format: safetensors
quantization: none
context_length: null
vram_required_gb: null
parameters: 4B
base_model: null
adapter_for: null
datasets: []
eval_suites: []
deployment:
target: docker-gpu.cin.su
runtime: transformers
served_model_name: lmt-60-4b
notes: "Apache-2.0 multilingual translation candidate. Chosen over NLLB for fewer license restrictions while staying below larger 8B translation models."
+32
View File
@@ -0,0 +1,32 @@
id: qwen-image-edit
name: Qwen Image Edit
type: image-diffusion-model
task:
- image-editing
- inpainting
- image-generation
language:
- ru
- en
- multilingual
source: huggingface
upstream_id: Qwen/Qwen-Image-Edit
upstream_url: https://huggingface.co/Qwen/Qwen-Image-Edit
license: apache-2.0
status: candidate
created_at: 2026-06-19
storage_path: /models/image/qwen-image-edit
format: diffusers
quantization: bf16
context_length: null
vram_required_gb: 24
parameters: 20B
base_model: Qwen/Qwen-Image
adapter_for: null
datasets: []
eval_suites: []
deployment:
target: docker-gpu.cin.su
runtime: transformers
served_model_name: qwen-image-edit
notes: "Heavy image editing candidate with stronger text/rendering/editing capabilities than SDXL. On 2026-06-20 it loaded as QwenImageEditPipeline in about 31s on RTX 4090 with CPU offload, but a 512x512 1-step edit did not finish within 1800s. Keep it as an experimental/manual-switch model until a quantized runtime or larger GPU is available."
@@ -0,0 +1,33 @@
id: qwen2_5-vl-7b-instruct
name: Qwen2.5-VL 7B Instruct
type: vision-language-model
task:
- video
- image-understanding
- document-understanding
- visual-question-answering
language:
- ru
- en
- multilingual
source: huggingface
upstream_id: Qwen/Qwen2.5-VL-7B-Instruct
upstream_url: https://huggingface.co/Qwen/Qwen2.5-VL-7B-Instruct
license: apache-2.0
status: candidate
created_at: 2026-06-19
storage_path: /models/video/qwen2.5-vl-7b-instruct
format: safetensors
quantization: none
context_length: null
vram_required_gb: null
parameters: 7B
base_model: Qwen/Qwen2.5-VL
adapter_for: null
datasets: []
eval_suites: []
deployment:
target: docker-gpu.cin.su
runtime: transformers
served_model_name: qwen2.5-vl-7b-instruct
notes: "Video/image understanding candidate for the video plugin. Supports long-video understanding according to the model card."
@@ -0,0 +1,38 @@
id: qwen3-14b-instruct-q6_k
name: Qwen3 14B Instruct GGUF Q6_K
type: gguf-model
task:
- text
- chat
- summarization
- code
- tool-use
- 1c-rag
language:
- ru
- en
- multilingual
source: huggingface
upstream_id: Qwen/Qwen3-14B-GGUF
upstream_url: https://huggingface.co/Qwen/Qwen3-14B-GGUF
license: apache-2.0
status: candidate
created_at: 2026-06-19
storage_path: /models/gguf/text/qwen3-14b-instruct-q6_k
filename: Qwen3-14B-Q6_K.gguf
file_size_bytes: 12121937248
format: gguf
quantization: Q6_K
context_length: 40960
deployment_context_length: 16384
vram_required_gb: 16
parameters: 14.8B
base_model: Qwen/Qwen3-14B
adapter_for: null
datasets: []
eval_suites: []
deployment:
target: docker-gpu.cin.su
runtime: llama.cpp
served_model_name: qwen3-14b-q6
notes: "Stronger general text/chat candidate than Qwen3 4B while still comfortable for a single RTX 4090 in GGUF Q6_K."
@@ -0,0 +1,32 @@
id: qwen3-4b-1c-lora-v1
name: Qwen3 4B 1C LoRA v1
type: lora-adapter
task:
- 1c
- bsl-code
- metadata-safety
- 1c-query
- explanation
language:
- ru
source: local-training
license: internal
status: draft
created_at: 2026-06-18
storage_path: /models/adapters/1c/qwen3-4b-1c-lora-v1
format: safetensors
quantization: null
context_length: 32768
vram_required_gb: null
parameters: null
base_model: qwen3-4b-instruct-2507
adapter_for: qwen3-4b-instruct-2507
datasets:
- plugins/1c/training/manifests/dataset.yaml
eval_suites:
- plugins/1c/evals/smoke.yaml
deployment:
target: docker-gpu.cin.su
runtime: vllm
served_model_name: qwen3-4b-1c
notes: "Draft 1C LoRA adapter. Smoke training completed on docker-gpu.cin.su, then repeated with a 65-record synthetic/example dataset generated from 1C metadata and BSL snapshots; use it to validate the training/deployment pipeline, not as a quality 1C expert model until a real reviewed dataset is prepared."
@@ -0,0 +1,37 @@
id: qwen3-4b-instruct-2507
name: Qwen3 4B Instruct 2507
type: base-model
task:
- text
- chat
- summarization
- code
- tool-use
- 1c-rag
language:
- ru
- en
source: huggingface
upstream_id: Qwen/Qwen3-4B-Instruct-2507
upstream_url: https://huggingface.co/Qwen/Qwen3-4B-Instruct-2507
license: apache-2.0
status: staging
created_at: 2026-06-18
storage_path: /models/base/qwen3-4b-instruct-2507
format: safetensors
quantization: none
context_length: 262144
deployment_context_length: 32768
vram_required_gb: null
parameters: 4B
base_model: null
adapter_for: null
datasets: []
eval_suites:
- evals/text/smoke.yaml
deployment:
target: docker-gpu.cin.su
runtime: vllm
served_model_name: qwen3-4b-instruct
compose: core/deploy/docker-gpu/vllm/compose.yaml
notes: "First base text model for local inference. Start with 32K context to reduce OOM risk; increase after GPU VRAM is confirmed."
@@ -0,0 +1,34 @@
id: qwen3-coder-30b-a3b-1c-lora-v1
name: Qwen3 Coder 30B A3B 1C LoRA v1
type: lora-adapter
task:
- 1c
- bsl-code
- metadata-safety
- metadata-write-verified
- metadata-write-learning-plan
- 1c-query
- explanation
language:
- ru
source: local-training
license: internal
status: draft
created_at: 2026-07-04
storage_path: /models/adapters/1c/qwen3-coder-30b-a3b-1c-lora-v1
format: safetensors
quantization: null
context_length: 262144
vram_required_gb: null
parameters: null
base_model: qwen3-coder-30b-a3b-instruct
adapter_for: qwen3-coder-30b-a3b-instruct
datasets:
- plugins/1c/training/manifests/dataset.yaml
eval_suites:
- plugins/1c/evals/smoke.yaml
deployment:
target: docker-gpu.cin.su
runtime: llama.cpp
served_model_name: qwen3-coder-1c-q6
notes: "Draft adapter for the current Qwen3-Coder Q6 route. Train against the HF base, then convert or merge for the llama.cpp GGUF deployment used on docker-gpu.cin.su."
@@ -0,0 +1,38 @@
id: qwen3-coder-30b-a3b-instruct-q4_k_m
name: Qwen3 Coder 30B A3B Instruct GGUF Q4_K_M
type: gguf-model
task:
- 1c
- code
- agentic-coding
- repository-analysis
- tool-use
language:
- ru
- en
source: huggingface
upstream_id: lmstudio-community/Qwen3-Coder-30B-A3B-Instruct-GGUF
upstream_url: https://huggingface.co/lmstudio-community/Qwen3-Coder-30B-A3B-Instruct-GGUF
license: apache-2.0
status: candidate
created_at: 2026-06-19
storage_path: /models/gguf/1c/qwen3-coder-30b-a3b-instruct-q4_k_m
filename: Qwen3-Coder-30B-A3B-Instruct-Q4_K_M.gguf
file_size_bytes: 18632186176
format: gguf
quantization: Q4_K_M
context_length: 262144
deployment_context_length: 32768
vram_required_gb: null
parameters: 30.5B total / 3.3B active
base_model: Qwen/Qwen3-Coder-30B-A3B-Instruct
adapter_for: null
datasets: []
eval_suites:
- plugins/1c/evals/smoke.yaml
deployment:
target: docker-gpu.cin.su
runtime: llama.cpp
served_model_name: qwen3-coder-1c-q4
compose: core/deploy/docker-gpu/llama-cpp/compose.yaml
notes: "Best current candidate found for 1C/code experiments. Not 1C-specific, but strong for agentic coding, long context, repository-scale understanding, and tool workflows."
@@ -0,0 +1,38 @@
id: qwen3-coder-30b-a3b-instruct-q6_k
name: Qwen3 Coder 30B A3B Instruct GGUF Q6_K
type: gguf-model
task:
- 1c
- code
- agentic-coding
- repository-analysis
- tool-use
language:
- ru
- en
source: huggingface
upstream_id: lmstudio-community/Qwen3-Coder-30B-A3B-Instruct-GGUF
upstream_url: https://huggingface.co/lmstudio-community/Qwen3-Coder-30B-A3B-Instruct-GGUF
license: apache-2.0
status: candidate
created_at: 2026-06-19
storage_path: /models/gguf/1c/qwen3-coder-30b-a3b-instruct-q6_k
filename: Qwen3-Coder-30B-A3B-Instruct-Q6_K.gguf
file_size_bytes: 25104724288
format: gguf
quantization: Q6_K
context_length: 262144
deployment_context_length: 8192
vram_required_gb: null
parameters: 30.5B total / 3.3B active
base_model: Qwen/Qwen3-Coder-30B-A3B-Instruct
adapter_for: null
datasets: []
eval_suites:
- plugins/1c/evals/smoke.yaml
deployment:
target: docker-gpu.cin.su
runtime: llama.cpp
served_model_name: qwen3-coder-1c-q6
compose: core/deploy/docker-gpu/llama-cpp/compose.yaml
notes: "High-quality Q6_K quant for local 1C/code experiments on a single RTX 4090. Expected to require partial CPU/RAM offload or a short context because the GGUF file is larger than 24 GiB VRAM. Test first at 8K context, then 16K if stable."
@@ -0,0 +1,35 @@
id: qwen3-coder-30b-a3b-instruct
name: Qwen3 Coder 30B A3B Instruct
type: base-model
task:
- 1c
- code
- agentic-coding
- repository-analysis
- tool-use
language:
- ru
- en
source: huggingface
upstream_id: Qwen/Qwen3-Coder-30B-A3B-Instruct
upstream_url: https://huggingface.co/Qwen/Qwen3-Coder-30B-A3B-Instruct
license: apache-2.0
status: staging
created_at: 2026-07-04
storage_path: /models/base/qwen3-coder-30b-a3b-instruct
format: safetensors
quantization: none
context_length: 262144
deployment_context_length: 8192
vram_required_gb: null
parameters: 30.5B total / 3.3B active
base_model: null
adapter_for: null
datasets: []
eval_suites:
- plugins/1c/evals/smoke.yaml
deployment:
target: docker-gpu.cin.su
runtime: training
served_model_name: qwen3-coder-30b-a3b-instruct
notes: "Training base for the current local 1C/code route. Use this HF checkpoint for LoRA fine-tuning, then convert or merge adapters for the Q6 GGUF deployment."
@@ -0,0 +1,33 @@
id: qwen3-vl-8b-instruct
name: Qwen3-VL 8B Instruct
type: vision-language-model
task:
- video
- image-understanding
- document-understanding
- visual-question-answering
language:
- ru
- en
- multilingual
source: huggingface
upstream_id: Qwen/Qwen3-VL-8B-Instruct
upstream_url: https://huggingface.co/Qwen/Qwen3-VL-8B-Instruct
license: apache-2.0
status: candidate
created_at: 2026-06-19
storage_path: /models/video/qwen3-vl-8b-instruct
format: safetensors
quantization: none
context_length: null
vram_required_gb: 18
parameters: 8B
base_model: Qwen/Qwen3-VL
adapter_for: null
datasets: []
eval_suites: []
deployment:
target: docker-gpu.cin.su
runtime: transformers
served_model_name: qwen3-vl-8b-instruct
notes: "Newer vision-language candidate for image, document, and video understanding tests. Should be tested separately from large text/image models."
+29
View File
@@ -0,0 +1,29 @@
id: realvisxl-v5_0
name: RealVisXL V5.0
type: image-diffusion-model
task:
- image-generation
language:
- en
- multilingual
source: huggingface
upstream_id: SG161222/RealVisXL_V5.0
upstream_url: https://huggingface.co/SG161222/RealVisXL_V5.0
license: openrail++
status: candidate
created_at: 2026-07-04
storage_path: /models/image/realvisxl-v5.0
format: diffusers
quantization: fp16
context_length: null
vram_required_gb: 12
parameters: null
base_model: stabilityai/stable-diffusion-xl-base-1.0
adapter_for: null
datasets: []
eval_suites: []
deployment:
target: docker-gpu.cin.su
runtime: transformers
served_model_name: realvisxl-v5
notes: "Photorealistic SDXL finetune for local RTX 4090 image generation. The Hugging Face model card exposes Diffusers usage and targets photorealistic output."
+29
View File
@@ -0,0 +1,29 @@
id: sdxl-base-1_0
name: Stable Diffusion XL Base 1.0
type: image-diffusion-model
task:
- image-generation
language:
- en
- multilingual
source: huggingface
upstream_id: stabilityai/stable-diffusion-xl-base-1.0
upstream_url: https://huggingface.co/stabilityai/stable-diffusion-xl-base-1.0
license: openrail++
status: staging
created_at: 2026-06-19
storage_path: /models/image/sdxl-base-1.0
format: diffusers
quantization: fp16
context_length: null
vram_required_gb: 12
parameters: null
base_model: null
adapter_for: null
datasets: []
eval_suites: []
deployment:
target: docker-gpu.cin.su
runtime: transformers
served_model_name: sdxl-image
notes: "Diffusers SDXL text-to-image candidate for local photo generation on RTX 4090."
@@ -0,0 +1,30 @@
id: sdxl-inpainting-1_0
name: Stable Diffusion XL Inpainting 1.0
type: image-diffusion-model
task:
- image-editing
- inpainting
language:
- en
- multilingual
source: huggingface
upstream_id: diffusers/stable-diffusion-xl-1.0-inpainting-0.1
upstream_url: https://huggingface.co/diffusers/stable-diffusion-xl-1.0-inpainting-0.1
license: openrail++
status: staging
created_at: 2026-06-19
storage_path: /models/image/sdxl-inpainting-1.0
format: diffusers
quantization: fp16
context_length: null
vram_required_gb: 12
parameters: null
base_model: stabilityai/stable-diffusion-xl-base-1.0
adapter_for: null
datasets: []
eval_suites: []
deployment:
target: docker-gpu.cin.su
runtime: transformers
served_model_name: sdxl-image
notes: "Diffusers SDXL inpainting candidate for local masked image editing."
@@ -0,0 +1,32 @@
id: whisper-large-v3-turbo-russian
name: Whisper Large v3 Turbo Russian
type: speech-model
task:
- speech-to-text
- speech-translation
language:
- ru
source: huggingface
upstream_id: dvislobokov/whisper-large-v3-turbo-russian
upstream_url: https://huggingface.co/dvislobokov/whisper-large-v3-turbo-russian
license: mit
status: candidate
created_at: 2026-06-19
storage_path: /models/audio/whisper-large-v3-turbo-russian
format: safetensors
quantization: none
filename: model.safetensors
file_size_bytes: 3235581408
context_length: null
vram_required_gb: 6
parameters: null
base_model: openai/whisper-large-v3-turbo
adapter_for: null
datasets:
- Mozilla Common Voice 17
eval_suites: []
deployment:
target: docker-gpu.cin.su
runtime: transformers
served_model_name: whisper-large-v3-turbo-russian
notes: "Russian fine-tuned Whisper turbo candidate for comparing Russian ASR quality against the base multilingual Whisper turbo."
@@ -0,0 +1,33 @@
id: whisper-large-v3-turbo
name: Whisper Large v3 Turbo
type: speech-model
task:
- speech-to-text
- speech-translation
language:
- ru
- en
- multilingual
source: huggingface
upstream_id: openai/whisper-large-v3-turbo
upstream_url: https://huggingface.co/openai/whisper-large-v3-turbo
license: mit
status: candidate
created_at: 2026-06-19
storage_path: /models/audio/whisper-large-v3-turbo
format: safetensors
quantization: none
filename: model.safetensors
file_size_bytes: 1617824864
context_length: null
vram_required_gb: null
parameters: null
base_model: openai/whisper-large-v3
adapter_for: null
datasets: []
eval_suites: []
deployment:
target: docker-gpu.cin.su
runtime: transformers
served_model_name: whisper-large-v3-turbo
notes: "ASR and speech translation model for the audio plugin."
+26
View File
@@ -0,0 +1,26 @@
id: example-model
name: Example Model
type: base-model
task:
- text
language:
- ru
- en
source: local
license: unknown
status: draft
created_at: 2026-06-18
storage_path: /models/base/example-model
format: safetensors
quantization: none
context_length: null
vram_required_gb: null
parameters: null
base_model: null
adapter_for: null
datasets: []
eval_suites: []
deployment:
target: docker-gpu.cin.su
runtime: null
notes: ""