llama3_3b_instruct_vallina_full_sft_30k
phi2-text-to-sql-full-20k
Llama3.1-8B-Math-v2
stablejack-0.5b-poc
day1-train-model
qwen-32B-extreme-sports-2
qwen-32B-bad-medical-dense-checkpoints
Qwen2.5-7B-Instruct-layers-1-10-smaller-lr
Alfred-ToRevuelto-1.5B
a1-github_dockerfiles
a1-toolscale
dare-model-0.1
GIM-1.7B
udk-ue3-qw34b-v2
udk-ue3-qw34b-v4
turkish-finance-qwen3b
toolcalling-merged-demo
code-grpo-checkpoint-200
code-grpo-checkpoint-800
Qwen2.5-0.5B
FAME_GD_llama32-3b-instruct-qa
FAME_PO_llama32-3b-instruct-qa
FAME-topics_PO_llama32-1b-instruct-qa
FAME-topics_GD_llama32-3b-instruct-qa
grpo-qwen-gsm8k
qwen2.5-14b-tensopolis-v1
model_sft_dare_0.9
model_sft_resta
ds1p5b_all-global_step_200
ds1p5b_no_if-global_step_200
finance-lora-qwen3-4b-merged
retrosynthesis-qwen3-4b
Mistral-7B-Erebus-v3
qwen2.5-7B-rlvr_g8_b384_math
model_sft_dare
model_dare_0.1
SearchR1-nq_hotpotqa_train-qwen2.5-3b-it-em-grpo-v0.2
Llama-Legal-Expression-8B-v0.1-merged
GRPO-non-thinking
Qwen3-0.6B-TL-SynthDolly-1A-E3
Qwen3-4B-Base-ascii-art-v6-joint-e3-neftune