curl --request POST \
--url https://api.nugen.in/api/v3/alignment-projects/{alignment_id}/auto-align \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"target_score": 88,
"benchmark_id": "<string>",
"method": "auto",
"representations": "auto",
"sequencing": [
"<string>"
],
"max_compute_hours": 123,
"notify_on_phase_transition": true,
"config": {
"adapter": {
"alpha": 32,
"dropout": 0,
"init_seed": 7,
"mode": "lora",
"quantization_bits": 4,
"rank": 16,
"train_attn": true,
"train_mlp": true,
"train_unembed": true
},
"advantage": {
"center": "group_mean",
"estimator": "gae",
"gae_lambda": 0.95,
"gamma": 0.99,
"normalize_std": true,
"scope": "per_token",
"skip_zero_advantage_groups": true
},
"base_model": "Qwen/Qwen2.5-0.5B-Instruct",
"checkpointing": {
"keep_last_n": 3,
"mode": "training",
"promote_final": true,
"save_every_steps": 50
},
"dataset": {
"answer_field": "responses",
"file_type": "jsonl",
"format": "chat",
"prompt_field": "instructions",
"shuffle": true,
"split": "train",
"streaming": true,
"synthetic": {
"enabled": false
}
},
"environment": {
"fail_fast": false,
"max_turns": 8,
"num_agents": 1,
"retry_on_failure": true,
"sandbox": {
"backend": "local",
"enabled": false
},
"tools": [],
"type": "multi_turn"
},
"evaluation": {
"eval_batch_size": 64,
"eval_every_steps": 50,
"eval_group_size": 1,
"eval_temperature": 0,
"metrics": [
"reward",
"task_success"
]
},
"logging": {
"log_metrics": [
"reward",
"loss",
"kl",
"entropy",
"grad_norm"
],
"wandb_project": "alignment-id"
},
"loss": {
"cispo": {
"eps_max": 6
},
"custom_weights_field": "token_weights",
"dro": {
"beta": 0.05
},
"entropy_coef": 0.001,
"is_ratio_level": "token",
"kl_coef": 0.02,
"ppo": {
"clip_high": 0.2,
"clip_low": 0.2
},
"token_reduction": "sum",
"type": "cispo",
"z_loss_coef": 0.0001
},
"optimizer": {
"beta1": 0.9,
"beta2": 0.95,
"eps": 1e-8,
"grad_clip_norm": 1,
"learning_rate": 0.000025,
"lr_schedule": "cosine",
"type": "adamw",
"warmup_steps": 20,
"weight_decay": 0
},
"reference_policy": {
"ema_decay": 0.999,
"enabled": true,
"kl_target": 0.05,
"source": "frozen_snapshot"
},
"rendering": {
"add_special_tokens": true,
"batch_unit": "tokens",
"max_seq_len": 65536,
"packing": true,
"truncation": "right"
},
"reward": {
"clip": [
-1,
1
],
"composite": {
"task_success": 1,
"tool_format": 0.2
},
"llm_judge": {
"criteria": "Were the right words used to reach the correct answer?",
"judge_model": "Qwen/Qwen2.5-0.5B-Instruct"
},
"normalize": "per_group_zscore",
"type": "llm_judge"
},
"sampling": {
"enable_thinking": true,
"group_size": 8,
"max_prompt_tokens": 4096,
"max_seq_len": 131072,
"max_tokens": 1024,
"min_p": 0,
"prompt_caching": true,
"prompt_groups_per_step": 64,
"seed": 12345,
"session_affinity_key": "trajectory_id",
"stop": [
"<|im_end|>"
],
"temperature": 1,
"thinking_effort": "high",
"top_k": 40,
"top_p": 0.95
},
"training": {
"batch_size": 256,
"early_stopping": {
"enabled": true,
"metric": "eval_loss",
"patience": 3
},
"epochs": 3,
"gradient_accumulation": 1,
"inner_optim_steps": 2,
"max_steps": 2000,
"mode": "async",
"rollout_training_overlap": true,
"seed": 42,
"steps": 500
}
}
}
'import requests
url = "https://api.nugen.in/api/v3/alignment-projects/{alignment_id}/auto-align"
payload = {
"target_score": 88,
"benchmark_id": "<string>",
"method": "auto",
"representations": "auto",
"sequencing": ["<string>"],
"max_compute_hours": 123,
"notify_on_phase_transition": True,
"config": {
"adapter": {
"alpha": 32,
"dropout": 0,
"init_seed": 7,
"mode": "lora",
"quantization_bits": 4,
"rank": 16,
"train_attn": True,
"train_mlp": True,
"train_unembed": True
},
"advantage": {
"center": "group_mean",
"estimator": "gae",
"gae_lambda": 0.95,
"gamma": 0.99,
"normalize_std": True,
"scope": "per_token",
"skip_zero_advantage_groups": True
},
"base_model": "Qwen/Qwen2.5-0.5B-Instruct",
"checkpointing": {
"keep_last_n": 3,
"mode": "training",
"promote_final": True,
"save_every_steps": 50
},
"dataset": {
"answer_field": "responses",
"file_type": "jsonl",
"format": "chat",
"prompt_field": "instructions",
"shuffle": True,
"split": "train",
"streaming": True,
"synthetic": { "enabled": False }
},
"environment": {
"fail_fast": False,
"max_turns": 8,
"num_agents": 1,
"retry_on_failure": True,
"sandbox": {
"backend": "local",
"enabled": False
},
"tools": [],
"type": "multi_turn"
},
"evaluation": {
"eval_batch_size": 64,
"eval_every_steps": 50,
"eval_group_size": 1,
"eval_temperature": 0,
"metrics": ["reward", "task_success"]
},
"logging": {
"log_metrics": ["reward", "loss", "kl", "entropy", "grad_norm"],
"wandb_project": "alignment-id"
},
"loss": {
"cispo": { "eps_max": 6 },
"custom_weights_field": "token_weights",
"dro": { "beta": 0.05 },
"entropy_coef": 0.001,
"is_ratio_level": "token",
"kl_coef": 0.02,
"ppo": {
"clip_high": 0.2,
"clip_low": 0.2
},
"token_reduction": "sum",
"type": "cispo",
"z_loss_coef": 0.0001
},
"optimizer": {
"beta1": 0.9,
"beta2": 0.95,
"eps": 1e-8,
"grad_clip_norm": 1,
"learning_rate": 0.000025,
"lr_schedule": "cosine",
"type": "adamw",
"warmup_steps": 20,
"weight_decay": 0
},
"reference_policy": {
"ema_decay": 0.999,
"enabled": True,
"kl_target": 0.05,
"source": "frozen_snapshot"
},
"rendering": {
"add_special_tokens": True,
"batch_unit": "tokens",
"max_seq_len": 65536,
"packing": True,
"truncation": "right"
},
"reward": {
"clip": [-1, 1],
"composite": {
"task_success": 1,
"tool_format": 0.2
},
"llm_judge": {
"criteria": "Were the right words used to reach the correct answer?",
"judge_model": "Qwen/Qwen2.5-0.5B-Instruct"
},
"normalize": "per_group_zscore",
"type": "llm_judge"
},
"sampling": {
"enable_thinking": True,
"group_size": 8,
"max_prompt_tokens": 4096,
"max_seq_len": 131072,
"max_tokens": 1024,
"min_p": 0,
"prompt_caching": True,
"prompt_groups_per_step": 64,
"seed": 12345,
"session_affinity_key": "trajectory_id",
"stop": ["<|im_end|>"],
"temperature": 1,
"thinking_effort": "high",
"top_k": 40,
"top_p": 0.95
},
"training": {
"batch_size": 256,
"early_stopping": {
"enabled": True,
"metric": "eval_loss",
"patience": 3
},
"epochs": 3,
"gradient_accumulation": 1,
"inner_optim_steps": 2,
"max_steps": 2000,
"mode": "async",
"rollout_training_overlap": True,
"seed": 42,
"steps": 500
}
}
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
target_score: 88,
benchmark_id: '<string>',
method: 'auto',
representations: 'auto',
sequencing: ['<string>'],
max_compute_hours: 123,
notify_on_phase_transition: true,
config: {
adapter: {
alpha: 32,
dropout: 0,
init_seed: 7,
mode: 'lora',
quantization_bits: 4,
rank: 16,
train_attn: true,
train_mlp: true,
train_unembed: true
},
advantage: {
center: 'group_mean',
estimator: 'gae',
gae_lambda: 0.95,
gamma: 0.99,
normalize_std: true,
scope: 'per_token',
skip_zero_advantage_groups: true
},
base_model: 'Qwen/Qwen2.5-0.5B-Instruct',
checkpointing: {keep_last_n: 3, mode: 'training', promote_final: true, save_every_steps: 50},
dataset: {
answer_field: 'responses',
file_type: 'jsonl',
format: 'chat',
prompt_field: 'instructions',
shuffle: true,
split: 'train',
streaming: true,
synthetic: {enabled: false}
},
environment: {
fail_fast: false,
max_turns: 8,
num_agents: 1,
retry_on_failure: true,
sandbox: {backend: 'local', enabled: false},
tools: [],
type: 'multi_turn'
},
evaluation: {
eval_batch_size: 64,
eval_every_steps: 50,
eval_group_size: 1,
eval_temperature: 0,
metrics: ['reward', 'task_success']
},
logging: {
log_metrics: ['reward', 'loss', 'kl', 'entropy', 'grad_norm'],
wandb_project: 'alignment-id'
},
loss: {
cispo: {eps_max: 6},
custom_weights_field: 'token_weights',
dro: {beta: 0.05},
entropy_coef: 0.001,
is_ratio_level: 'token',
kl_coef: 0.02,
ppo: {clip_high: 0.2, clip_low: 0.2},
token_reduction: 'sum',
type: 'cispo',
z_loss_coef: 0.0001
},
optimizer: {
beta1: 0.9,
beta2: 0.95,
eps: 1e-8,
grad_clip_norm: 1,
learning_rate: 0.000025,
lr_schedule: 'cosine',
type: 'adamw',
warmup_steps: 20,
weight_decay: 0
},
reference_policy: {ema_decay: 0.999, enabled: true, kl_target: 0.05, source: 'frozen_snapshot'},
rendering: {
add_special_tokens: true,
batch_unit: 'tokens',
max_seq_len: 65536,
packing: true,
truncation: 'right'
},
reward: {
clip: [-1, 1],
composite: {task_success: 1, tool_format: 0.2},
llm_judge: {
criteria: 'Were the right words used to reach the correct answer?',
judge_model: 'Qwen/Qwen2.5-0.5B-Instruct'
},
normalize: 'per_group_zscore',
type: 'llm_judge'
},
sampling: {
enable_thinking: true,
group_size: 8,
max_prompt_tokens: 4096,
max_seq_len: 131072,
max_tokens: 1024,
min_p: 0,
prompt_caching: true,
prompt_groups_per_step: 64,
seed: 12345,
session_affinity_key: 'trajectory_id',
stop: ['<|im_end|>'],
temperature: 1,
thinking_effort: 'high',
top_k: 40,
top_p: 0.95
},
training: {
batch_size: 256,
early_stopping: {enabled: true, metric: 'eval_loss', patience: 3},
epochs: 3,
gradient_accumulation: 1,
inner_optim_steps: 2,
max_steps: 2000,
mode: 'async',
rollout_training_overlap: true,
seed: 42,
steps: 500
}
}
})
};
fetch('https://api.nugen.in/api/v3/alignment-projects/{alignment_id}/auto-align', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.nugen.in/api/v3/alignment-projects/{alignment_id}/auto-align",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'target_score' => 88,
'benchmark_id' => '<string>',
'method' => 'auto',
'representations' => 'auto',
'sequencing' => [
'<string>'
],
'max_compute_hours' => 123,
'notify_on_phase_transition' => true,
'config' => [
'adapter' => [
'alpha' => 32,
'dropout' => 0,
'init_seed' => 7,
'mode' => 'lora',
'quantization_bits' => 4,
'rank' => 16,
'train_attn' => true,
'train_mlp' => true,
'train_unembed' => true
],
'advantage' => [
'center' => 'group_mean',
'estimator' => 'gae',
'gae_lambda' => 0.95,
'gamma' => 0.99,
'normalize_std' => true,
'scope' => 'per_token',
'skip_zero_advantage_groups' => true
],
'base_model' => 'Qwen/Qwen2.5-0.5B-Instruct',
'checkpointing' => [
'keep_last_n' => 3,
'mode' => 'training',
'promote_final' => true,
'save_every_steps' => 50
],
'dataset' => [
'answer_field' => 'responses',
'file_type' => 'jsonl',
'format' => 'chat',
'prompt_field' => 'instructions',
'shuffle' => true,
'split' => 'train',
'streaming' => true,
'synthetic' => [
'enabled' => false
]
],
'environment' => [
'fail_fast' => false,
'max_turns' => 8,
'num_agents' => 1,
'retry_on_failure' => true,
'sandbox' => [
'backend' => 'local',
'enabled' => false
],
'tools' => [
],
'type' => 'multi_turn'
],
'evaluation' => [
'eval_batch_size' => 64,
'eval_every_steps' => 50,
'eval_group_size' => 1,
'eval_temperature' => 0,
'metrics' => [
'reward',
'task_success'
]
],
'logging' => [
'log_metrics' => [
'reward',
'loss',
'kl',
'entropy',
'grad_norm'
],
'wandb_project' => 'alignment-id'
],
'loss' => [
'cispo' => [
'eps_max' => 6
],
'custom_weights_field' => 'token_weights',
'dro' => [
'beta' => 0.05
],
'entropy_coef' => 0.001,
'is_ratio_level' => 'token',
'kl_coef' => 0.02,
'ppo' => [
'clip_high' => 0.2,
'clip_low' => 0.2
],
'token_reduction' => 'sum',
'type' => 'cispo',
'z_loss_coef' => 0.0001
],
'optimizer' => [
'beta1' => 0.9,
'beta2' => 0.95,
'eps' => 1e-8,
'grad_clip_norm' => 1,
'learning_rate' => 0.000025,
'lr_schedule' => 'cosine',
'type' => 'adamw',
'warmup_steps' => 20,
'weight_decay' => 0
],
'reference_policy' => [
'ema_decay' => 0.999,
'enabled' => true,
'kl_target' => 0.05,
'source' => 'frozen_snapshot'
],
'rendering' => [
'add_special_tokens' => true,
'batch_unit' => 'tokens',
'max_seq_len' => 65536,
'packing' => true,
'truncation' => 'right'
],
'reward' => [
'clip' => [
-1,
1
],
'composite' => [
'task_success' => 1,
'tool_format' => 0.2
],
'llm_judge' => [
'criteria' => 'Were the right words used to reach the correct answer?',
'judge_model' => 'Qwen/Qwen2.5-0.5B-Instruct'
],
'normalize' => 'per_group_zscore',
'type' => 'llm_judge'
],
'sampling' => [
'enable_thinking' => true,
'group_size' => 8,
'max_prompt_tokens' => 4096,
'max_seq_len' => 131072,
'max_tokens' => 1024,
'min_p' => 0,
'prompt_caching' => true,
'prompt_groups_per_step' => 64,
'seed' => 12345,
'session_affinity_key' => 'trajectory_id',
'stop' => [
'<|im_end|>'
],
'temperature' => 1,
'thinking_effort' => 'high',
'top_k' => 40,
'top_p' => 0.95
],
'training' => [
'batch_size' => 256,
'early_stopping' => [
'enabled' => true,
'metric' => 'eval_loss',
'patience' => 3
],
'epochs' => 3,
'gradient_accumulation' => 1,
'inner_optim_steps' => 2,
'max_steps' => 2000,
'mode' => 'async',
'rollout_training_overlap' => true,
'seed' => 42,
'steps' => 500
]
]
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.nugen.in/api/v3/alignment-projects/{alignment_id}/auto-align"
payload := strings.NewReader("{\n \"target_score\": 88,\n \"benchmark_id\": \"<string>\",\n \"method\": \"auto\",\n \"representations\": \"auto\",\n \"sequencing\": [\n \"<string>\"\n ],\n \"max_compute_hours\": 123,\n \"notify_on_phase_transition\": true,\n \"config\": {\n \"adapter\": {\n \"alpha\": 32,\n \"dropout\": 0,\n \"init_seed\": 7,\n \"mode\": \"lora\",\n \"quantization_bits\": 4,\n \"rank\": 16,\n \"train_attn\": true,\n \"train_mlp\": true,\n \"train_unembed\": true\n },\n \"advantage\": {\n \"center\": \"group_mean\",\n \"estimator\": \"gae\",\n \"gae_lambda\": 0.95,\n \"gamma\": 0.99,\n \"normalize_std\": true,\n \"scope\": \"per_token\",\n \"skip_zero_advantage_groups\": true\n },\n \"base_model\": \"Qwen/Qwen2.5-0.5B-Instruct\",\n \"checkpointing\": {\n \"keep_last_n\": 3,\n \"mode\": \"training\",\n \"promote_final\": true,\n \"save_every_steps\": 50\n },\n \"dataset\": {\n \"answer_field\": \"responses\",\n \"file_type\": \"jsonl\",\n \"format\": \"chat\",\n \"prompt_field\": \"instructions\",\n \"shuffle\": true,\n \"split\": \"train\",\n \"streaming\": true,\n \"synthetic\": {\n \"enabled\": false\n }\n },\n \"environment\": {\n \"fail_fast\": false,\n \"max_turns\": 8,\n \"num_agents\": 1,\n \"retry_on_failure\": true,\n \"sandbox\": {\n \"backend\": \"local\",\n \"enabled\": false\n },\n \"tools\": [],\n \"type\": \"multi_turn\"\n },\n \"evaluation\": {\n \"eval_batch_size\": 64,\n \"eval_every_steps\": 50,\n \"eval_group_size\": 1,\n \"eval_temperature\": 0,\n \"metrics\": [\n \"reward\",\n \"task_success\"\n ]\n },\n \"logging\": {\n \"log_metrics\": [\n \"reward\",\n \"loss\",\n \"kl\",\n \"entropy\",\n \"grad_norm\"\n ],\n \"wandb_project\": \"alignment-id\"\n },\n \"loss\": {\n \"cispo\": {\n \"eps_max\": 6\n },\n \"custom_weights_field\": \"token_weights\",\n \"dro\": {\n \"beta\": 0.05\n },\n \"entropy_coef\": 0.001,\n \"is_ratio_level\": \"token\",\n \"kl_coef\": 0.02,\n \"ppo\": {\n \"clip_high\": 0.2,\n \"clip_low\": 0.2\n },\n \"token_reduction\": \"sum\",\n \"type\": \"cispo\",\n \"z_loss_coef\": 0.0001\n },\n \"optimizer\": {\n \"beta1\": 0.9,\n \"beta2\": 0.95,\n \"eps\": 1e-8,\n \"grad_clip_norm\": 1,\n \"learning_rate\": 0.000025,\n \"lr_schedule\": \"cosine\",\n \"type\": \"adamw\",\n \"warmup_steps\": 20,\n \"weight_decay\": 0\n },\n \"reference_policy\": {\n \"ema_decay\": 0.999,\n \"enabled\": true,\n \"kl_target\": 0.05,\n \"source\": \"frozen_snapshot\"\n },\n \"rendering\": {\n \"add_special_tokens\": true,\n \"batch_unit\": \"tokens\",\n \"max_seq_len\": 65536,\n \"packing\": true,\n \"truncation\": \"right\"\n },\n \"reward\": {\n \"clip\": [\n -1,\n 1\n ],\n \"composite\": {\n \"task_success\": 1,\n \"tool_format\": 0.2\n },\n \"llm_judge\": {\n \"criteria\": \"Were the right words used to reach the correct answer?\",\n \"judge_model\": \"Qwen/Qwen2.5-0.5B-Instruct\"\n },\n \"normalize\": \"per_group_zscore\",\n \"type\": \"llm_judge\"\n },\n \"sampling\": {\n \"enable_thinking\": true,\n \"group_size\": 8,\n \"max_prompt_tokens\": 4096,\n \"max_seq_len\": 131072,\n \"max_tokens\": 1024,\n \"min_p\": 0,\n \"prompt_caching\": true,\n \"prompt_groups_per_step\": 64,\n \"seed\": 12345,\n \"session_affinity_key\": \"trajectory_id\",\n \"stop\": [\n \"<|im_end|>\"\n ],\n \"temperature\": 1,\n \"thinking_effort\": \"high\",\n \"top_k\": 40,\n \"top_p\": 0.95\n },\n \"training\": {\n \"batch_size\": 256,\n \"early_stopping\": {\n \"enabled\": true,\n \"metric\": \"eval_loss\",\n \"patience\": 3\n },\n \"epochs\": 3,\n \"gradient_accumulation\": 1,\n \"inner_optim_steps\": 2,\n \"max_steps\": 2000,\n \"mode\": \"async\",\n \"rollout_training_overlap\": true,\n \"seed\": 42,\n \"steps\": 500\n }\n }\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.nugen.in/api/v3/alignment-projects/{alignment_id}/auto-align")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"target_score\": 88,\n \"benchmark_id\": \"<string>\",\n \"method\": \"auto\",\n \"representations\": \"auto\",\n \"sequencing\": [\n \"<string>\"\n ],\n \"max_compute_hours\": 123,\n \"notify_on_phase_transition\": true,\n \"config\": {\n \"adapter\": {\n \"alpha\": 32,\n \"dropout\": 0,\n \"init_seed\": 7,\n \"mode\": \"lora\",\n \"quantization_bits\": 4,\n \"rank\": 16,\n \"train_attn\": true,\n \"train_mlp\": true,\n \"train_unembed\": true\n },\n \"advantage\": {\n \"center\": \"group_mean\",\n \"estimator\": \"gae\",\n \"gae_lambda\": 0.95,\n \"gamma\": 0.99,\n \"normalize_std\": true,\n \"scope\": \"per_token\",\n \"skip_zero_advantage_groups\": true\n },\n \"base_model\": \"Qwen/Qwen2.5-0.5B-Instruct\",\n \"checkpointing\": {\n \"keep_last_n\": 3,\n \"mode\": \"training\",\n \"promote_final\": true,\n \"save_every_steps\": 50\n },\n \"dataset\": {\n \"answer_field\": \"responses\",\n \"file_type\": \"jsonl\",\n \"format\": \"chat\",\n \"prompt_field\": \"instructions\",\n \"shuffle\": true,\n \"split\": \"train\",\n \"streaming\": true,\n \"synthetic\": {\n \"enabled\": false\n }\n },\n \"environment\": {\n \"fail_fast\": false,\n \"max_turns\": 8,\n \"num_agents\": 1,\n \"retry_on_failure\": true,\n \"sandbox\": {\n \"backend\": \"local\",\n \"enabled\": false\n },\n \"tools\": [],\n \"type\": \"multi_turn\"\n },\n \"evaluation\": {\n \"eval_batch_size\": 64,\n \"eval_every_steps\": 50,\n \"eval_group_size\": 1,\n \"eval_temperature\": 0,\n \"metrics\": [\n \"reward\",\n \"task_success\"\n ]\n },\n \"logging\": {\n \"log_metrics\": [\n \"reward\",\n \"loss\",\n \"kl\",\n \"entropy\",\n \"grad_norm\"\n ],\n \"wandb_project\": \"alignment-id\"\n },\n \"loss\": {\n \"cispo\": {\n \"eps_max\": 6\n },\n \"custom_weights_field\": \"token_weights\",\n \"dro\": {\n \"beta\": 0.05\n },\n \"entropy_coef\": 0.001,\n \"is_ratio_level\": \"token\",\n \"kl_coef\": 0.02,\n \"ppo\": {\n \"clip_high\": 0.2,\n \"clip_low\": 0.2\n },\n \"token_reduction\": \"sum\",\n \"type\": \"cispo\",\n \"z_loss_coef\": 0.0001\n },\n \"optimizer\": {\n \"beta1\": 0.9,\n \"beta2\": 0.95,\n \"eps\": 1e-8,\n \"grad_clip_norm\": 1,\n \"learning_rate\": 0.000025,\n \"lr_schedule\": \"cosine\",\n \"type\": \"adamw\",\n \"warmup_steps\": 20,\n \"weight_decay\": 0\n },\n \"reference_policy\": {\n \"ema_decay\": 0.999,\n \"enabled\": true,\n \"kl_target\": 0.05,\n \"source\": \"frozen_snapshot\"\n },\n \"rendering\": {\n \"add_special_tokens\": true,\n \"batch_unit\": \"tokens\",\n \"max_seq_len\": 65536,\n \"packing\": true,\n \"truncation\": \"right\"\n },\n \"reward\": {\n \"clip\": [\n -1,\n 1\n ],\n \"composite\": {\n \"task_success\": 1,\n \"tool_format\": 0.2\n },\n \"llm_judge\": {\n \"criteria\": \"Were the right words used to reach the correct answer?\",\n \"judge_model\": \"Qwen/Qwen2.5-0.5B-Instruct\"\n },\n \"normalize\": \"per_group_zscore\",\n \"type\": \"llm_judge\"\n },\n \"sampling\": {\n \"enable_thinking\": true,\n \"group_size\": 8,\n \"max_prompt_tokens\": 4096,\n \"max_seq_len\": 131072,\n \"max_tokens\": 1024,\n \"min_p\": 0,\n \"prompt_caching\": true,\n \"prompt_groups_per_step\": 64,\n \"seed\": 12345,\n \"session_affinity_key\": \"trajectory_id\",\n \"stop\": [\n \"<|im_end|>\"\n ],\n \"temperature\": 1,\n \"thinking_effort\": \"high\",\n \"top_k\": 40,\n \"top_p\": 0.95\n },\n \"training\": {\n \"batch_size\": 256,\n \"early_stopping\": {\n \"enabled\": true,\n \"metric\": \"eval_loss\",\n \"patience\": 3\n },\n \"epochs\": 3,\n \"gradient_accumulation\": 1,\n \"inner_optim_steps\": 2,\n \"max_steps\": 2000,\n \"mode\": \"async\",\n \"rollout_training_overlap\": true,\n \"seed\": 42,\n \"steps\": 500\n }\n }\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.nugen.in/api/v3/alignment-projects/{alignment_id}/auto-align")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"target_score\": 88,\n \"benchmark_id\": \"<string>\",\n \"method\": \"auto\",\n \"representations\": \"auto\",\n \"sequencing\": [\n \"<string>\"\n ],\n \"max_compute_hours\": 123,\n \"notify_on_phase_transition\": true,\n \"config\": {\n \"adapter\": {\n \"alpha\": 32,\n \"dropout\": 0,\n \"init_seed\": 7,\n \"mode\": \"lora\",\n \"quantization_bits\": 4,\n \"rank\": 16,\n \"train_attn\": true,\n \"train_mlp\": true,\n \"train_unembed\": true\n },\n \"advantage\": {\n \"center\": \"group_mean\",\n \"estimator\": \"gae\",\n \"gae_lambda\": 0.95,\n \"gamma\": 0.99,\n \"normalize_std\": true,\n \"scope\": \"per_token\",\n \"skip_zero_advantage_groups\": true\n },\n \"base_model\": \"Qwen/Qwen2.5-0.5B-Instruct\",\n \"checkpointing\": {\n \"keep_last_n\": 3,\n \"mode\": \"training\",\n \"promote_final\": true,\n \"save_every_steps\": 50\n },\n \"dataset\": {\n \"answer_field\": \"responses\",\n \"file_type\": \"jsonl\",\n \"format\": \"chat\",\n \"prompt_field\": \"instructions\",\n \"shuffle\": true,\n \"split\": \"train\",\n \"streaming\": true,\n \"synthetic\": {\n \"enabled\": false\n }\n },\n \"environment\": {\n \"fail_fast\": false,\n \"max_turns\": 8,\n \"num_agents\": 1,\n \"retry_on_failure\": true,\n \"sandbox\": {\n \"backend\": \"local\",\n \"enabled\": false\n },\n \"tools\": [],\n \"type\": \"multi_turn\"\n },\n \"evaluation\": {\n \"eval_batch_size\": 64,\n \"eval_every_steps\": 50,\n \"eval_group_size\": 1,\n \"eval_temperature\": 0,\n \"metrics\": [\n \"reward\",\n \"task_success\"\n ]\n },\n \"logging\": {\n \"log_metrics\": [\n \"reward\",\n \"loss\",\n \"kl\",\n \"entropy\",\n \"grad_norm\"\n ],\n \"wandb_project\": \"alignment-id\"\n },\n \"loss\": {\n \"cispo\": {\n \"eps_max\": 6\n },\n \"custom_weights_field\": \"token_weights\",\n \"dro\": {\n \"beta\": 0.05\n },\n \"entropy_coef\": 0.001,\n \"is_ratio_level\": \"token\",\n \"kl_coef\": 0.02,\n \"ppo\": {\n \"clip_high\": 0.2,\n \"clip_low\": 0.2\n },\n \"token_reduction\": \"sum\",\n \"type\": \"cispo\",\n \"z_loss_coef\": 0.0001\n },\n \"optimizer\": {\n \"beta1\": 0.9,\n \"beta2\": 0.95,\n \"eps\": 1e-8,\n \"grad_clip_norm\": 1,\n \"learning_rate\": 0.000025,\n \"lr_schedule\": \"cosine\",\n \"type\": \"adamw\",\n \"warmup_steps\": 20,\n \"weight_decay\": 0\n },\n \"reference_policy\": {\n \"ema_decay\": 0.999,\n \"enabled\": true,\n \"kl_target\": 0.05,\n \"source\": \"frozen_snapshot\"\n },\n \"rendering\": {\n \"add_special_tokens\": true,\n \"batch_unit\": \"tokens\",\n \"max_seq_len\": 65536,\n \"packing\": true,\n \"truncation\": \"right\"\n },\n \"reward\": {\n \"clip\": [\n -1,\n 1\n ],\n \"composite\": {\n \"task_success\": 1,\n \"tool_format\": 0.2\n },\n \"llm_judge\": {\n \"criteria\": \"Were the right words used to reach the correct answer?\",\n \"judge_model\": \"Qwen/Qwen2.5-0.5B-Instruct\"\n },\n \"normalize\": \"per_group_zscore\",\n \"type\": \"llm_judge\"\n },\n \"sampling\": {\n \"enable_thinking\": true,\n \"group_size\": 8,\n \"max_prompt_tokens\": 4096,\n \"max_seq_len\": 131072,\n \"max_tokens\": 1024,\n \"min_p\": 0,\n \"prompt_caching\": true,\n \"prompt_groups_per_step\": 64,\n \"seed\": 12345,\n \"session_affinity_key\": \"trajectory_id\",\n \"stop\": [\n \"<|im_end|>\"\n ],\n \"temperature\": 1,\n \"thinking_effort\": \"high\",\n \"top_k\": 40,\n \"top_p\": 0.95\n },\n \"training\": {\n \"batch_size\": 256,\n \"early_stopping\": {\n \"enabled\": true,\n \"metric\": \"eval_loss\",\n \"patience\": 3\n },\n \"epochs\": 3,\n \"gradient_accumulation\": 1,\n \"inner_optim_steps\": 2,\n \"max_steps\": 2000,\n \"mode\": \"async\",\n \"rollout_training_overlap\": true,\n \"seed\": 42,\n \"steps\": 500\n }\n }\n}"
response = http.request(request)
puts response.read_body{
"auto_align_id": "<string>",
"alignment_id": "<string>",
"status": "<string>",
"message": "<string>"
}{
"detail": [
{
"loc": [
"<string>"
],
"msg": "<string>",
"type": "<string>"
}
]
}Start Auto-Align
Submit a target alignment score and let the system plan the run.
The system analyses your data, base model and prior train-time alignment history to build a multi-phase alignment plan.
Phases are sequenced and executed continuously. The system may apply gradient-free representation adjustments, supervised learning, reward-guided optimisation, or a combination, re-planning when the score trajectory demands it. Checkpoints are evaluated between phases, and the system revisits earlier phases if regressions surface.
Every parameter defaults to auto. You can override any of them to constrain
the search, and the system performs best when given room to explore.
Path Parameters:
alignment_id: The alignment project to auto-align. Its base model, data and prior train-time alignment history warm-start the planning.
Request Body:
target_score(required): Target quality on a 0-99 scalebenchmark_id(optional): Benchmark that measures the achieved score against the target. Defaults to the benchmark already on the projectmethod,representations,sequencing(optional): Constrain which optimisation families run and in what ordermax_compute_hours(optional): Upper bound on GPU hours across all phasesnotify_on_phase_transition(optional): Notify on each phase changeconfig(optional): Take control of any part of the run, for reinforcement learning or supervised (SFT) settings: loss, optimizer, LoRA adapter, reward, advantage, reference policy, environment, dataset, sampling, training, checkpointing, evaluation, rendering and logging. Pass only what you want to pin; anything omitted stays auto and redundant settings are discarded. Every field is listed in theAutoAlignConfigschema, with a full example
Returns:
auto_align_id: Identifier of this requestalignment_id: The project it targetsstatus,message
Raises:
404: If the alignment project is not found or does not belong to you
Notes:
- Auto-alignment is an Enterprise Edition capability. The request is recorded and our team follows up; raise a support ticket to enable it
- Once enabled, poll
GET /alignment-projects/{alignment_id}/statusfor progress, phase transitions and intermediate scores
curl --request POST \
--url https://api.nugen.in/api/v3/alignment-projects/{alignment_id}/auto-align \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"target_score": 88,
"benchmark_id": "<string>",
"method": "auto",
"representations": "auto",
"sequencing": [
"<string>"
],
"max_compute_hours": 123,
"notify_on_phase_transition": true,
"config": {
"adapter": {
"alpha": 32,
"dropout": 0,
"init_seed": 7,
"mode": "lora",
"quantization_bits": 4,
"rank": 16,
"train_attn": true,
"train_mlp": true,
"train_unembed": true
},
"advantage": {
"center": "group_mean",
"estimator": "gae",
"gae_lambda": 0.95,
"gamma": 0.99,
"normalize_std": true,
"scope": "per_token",
"skip_zero_advantage_groups": true
},
"base_model": "Qwen/Qwen2.5-0.5B-Instruct",
"checkpointing": {
"keep_last_n": 3,
"mode": "training",
"promote_final": true,
"save_every_steps": 50
},
"dataset": {
"answer_field": "responses",
"file_type": "jsonl",
"format": "chat",
"prompt_field": "instructions",
"shuffle": true,
"split": "train",
"streaming": true,
"synthetic": {
"enabled": false
}
},
"environment": {
"fail_fast": false,
"max_turns": 8,
"num_agents": 1,
"retry_on_failure": true,
"sandbox": {
"backend": "local",
"enabled": false
},
"tools": [],
"type": "multi_turn"
},
"evaluation": {
"eval_batch_size": 64,
"eval_every_steps": 50,
"eval_group_size": 1,
"eval_temperature": 0,
"metrics": [
"reward",
"task_success"
]
},
"logging": {
"log_metrics": [
"reward",
"loss",
"kl",
"entropy",
"grad_norm"
],
"wandb_project": "alignment-id"
},
"loss": {
"cispo": {
"eps_max": 6
},
"custom_weights_field": "token_weights",
"dro": {
"beta": 0.05
},
"entropy_coef": 0.001,
"is_ratio_level": "token",
"kl_coef": 0.02,
"ppo": {
"clip_high": 0.2,
"clip_low": 0.2
},
"token_reduction": "sum",
"type": "cispo",
"z_loss_coef": 0.0001
},
"optimizer": {
"beta1": 0.9,
"beta2": 0.95,
"eps": 1e-8,
"grad_clip_norm": 1,
"learning_rate": 0.000025,
"lr_schedule": "cosine",
"type": "adamw",
"warmup_steps": 20,
"weight_decay": 0
},
"reference_policy": {
"ema_decay": 0.999,
"enabled": true,
"kl_target": 0.05,
"source": "frozen_snapshot"
},
"rendering": {
"add_special_tokens": true,
"batch_unit": "tokens",
"max_seq_len": 65536,
"packing": true,
"truncation": "right"
},
"reward": {
"clip": [
-1,
1
],
"composite": {
"task_success": 1,
"tool_format": 0.2
},
"llm_judge": {
"criteria": "Were the right words used to reach the correct answer?",
"judge_model": "Qwen/Qwen2.5-0.5B-Instruct"
},
"normalize": "per_group_zscore",
"type": "llm_judge"
},
"sampling": {
"enable_thinking": true,
"group_size": 8,
"max_prompt_tokens": 4096,
"max_seq_len": 131072,
"max_tokens": 1024,
"min_p": 0,
"prompt_caching": true,
"prompt_groups_per_step": 64,
"seed": 12345,
"session_affinity_key": "trajectory_id",
"stop": [
"<|im_end|>"
],
"temperature": 1,
"thinking_effort": "high",
"top_k": 40,
"top_p": 0.95
},
"training": {
"batch_size": 256,
"early_stopping": {
"enabled": true,
"metric": "eval_loss",
"patience": 3
},
"epochs": 3,
"gradient_accumulation": 1,
"inner_optim_steps": 2,
"max_steps": 2000,
"mode": "async",
"rollout_training_overlap": true,
"seed": 42,
"steps": 500
}
}
}
'import requests
url = "https://api.nugen.in/api/v3/alignment-projects/{alignment_id}/auto-align"
payload = {
"target_score": 88,
"benchmark_id": "<string>",
"method": "auto",
"representations": "auto",
"sequencing": ["<string>"],
"max_compute_hours": 123,
"notify_on_phase_transition": True,
"config": {
"adapter": {
"alpha": 32,
"dropout": 0,
"init_seed": 7,
"mode": "lora",
"quantization_bits": 4,
"rank": 16,
"train_attn": True,
"train_mlp": True,
"train_unembed": True
},
"advantage": {
"center": "group_mean",
"estimator": "gae",
"gae_lambda": 0.95,
"gamma": 0.99,
"normalize_std": True,
"scope": "per_token",
"skip_zero_advantage_groups": True
},
"base_model": "Qwen/Qwen2.5-0.5B-Instruct",
"checkpointing": {
"keep_last_n": 3,
"mode": "training",
"promote_final": True,
"save_every_steps": 50
},
"dataset": {
"answer_field": "responses",
"file_type": "jsonl",
"format": "chat",
"prompt_field": "instructions",
"shuffle": True,
"split": "train",
"streaming": True,
"synthetic": { "enabled": False }
},
"environment": {
"fail_fast": False,
"max_turns": 8,
"num_agents": 1,
"retry_on_failure": True,
"sandbox": {
"backend": "local",
"enabled": False
},
"tools": [],
"type": "multi_turn"
},
"evaluation": {
"eval_batch_size": 64,
"eval_every_steps": 50,
"eval_group_size": 1,
"eval_temperature": 0,
"metrics": ["reward", "task_success"]
},
"logging": {
"log_metrics": ["reward", "loss", "kl", "entropy", "grad_norm"],
"wandb_project": "alignment-id"
},
"loss": {
"cispo": { "eps_max": 6 },
"custom_weights_field": "token_weights",
"dro": { "beta": 0.05 },
"entropy_coef": 0.001,
"is_ratio_level": "token",
"kl_coef": 0.02,
"ppo": {
"clip_high": 0.2,
"clip_low": 0.2
},
"token_reduction": "sum",
"type": "cispo",
"z_loss_coef": 0.0001
},
"optimizer": {
"beta1": 0.9,
"beta2": 0.95,
"eps": 1e-8,
"grad_clip_norm": 1,
"learning_rate": 0.000025,
"lr_schedule": "cosine",
"type": "adamw",
"warmup_steps": 20,
"weight_decay": 0
},
"reference_policy": {
"ema_decay": 0.999,
"enabled": True,
"kl_target": 0.05,
"source": "frozen_snapshot"
},
"rendering": {
"add_special_tokens": True,
"batch_unit": "tokens",
"max_seq_len": 65536,
"packing": True,
"truncation": "right"
},
"reward": {
"clip": [-1, 1],
"composite": {
"task_success": 1,
"tool_format": 0.2
},
"llm_judge": {
"criteria": "Were the right words used to reach the correct answer?",
"judge_model": "Qwen/Qwen2.5-0.5B-Instruct"
},
"normalize": "per_group_zscore",
"type": "llm_judge"
},
"sampling": {
"enable_thinking": True,
"group_size": 8,
"max_prompt_tokens": 4096,
"max_seq_len": 131072,
"max_tokens": 1024,
"min_p": 0,
"prompt_caching": True,
"prompt_groups_per_step": 64,
"seed": 12345,
"session_affinity_key": "trajectory_id",
"stop": ["<|im_end|>"],
"temperature": 1,
"thinking_effort": "high",
"top_k": 40,
"top_p": 0.95
},
"training": {
"batch_size": 256,
"early_stopping": {
"enabled": True,
"metric": "eval_loss",
"patience": 3
},
"epochs": 3,
"gradient_accumulation": 1,
"inner_optim_steps": 2,
"max_steps": 2000,
"mode": "async",
"rollout_training_overlap": True,
"seed": 42,
"steps": 500
}
}
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
target_score: 88,
benchmark_id: '<string>',
method: 'auto',
representations: 'auto',
sequencing: ['<string>'],
max_compute_hours: 123,
notify_on_phase_transition: true,
config: {
adapter: {
alpha: 32,
dropout: 0,
init_seed: 7,
mode: 'lora',
quantization_bits: 4,
rank: 16,
train_attn: true,
train_mlp: true,
train_unembed: true
},
advantage: {
center: 'group_mean',
estimator: 'gae',
gae_lambda: 0.95,
gamma: 0.99,
normalize_std: true,
scope: 'per_token',
skip_zero_advantage_groups: true
},
base_model: 'Qwen/Qwen2.5-0.5B-Instruct',
checkpointing: {keep_last_n: 3, mode: 'training', promote_final: true, save_every_steps: 50},
dataset: {
answer_field: 'responses',
file_type: 'jsonl',
format: 'chat',
prompt_field: 'instructions',
shuffle: true,
split: 'train',
streaming: true,
synthetic: {enabled: false}
},
environment: {
fail_fast: false,
max_turns: 8,
num_agents: 1,
retry_on_failure: true,
sandbox: {backend: 'local', enabled: false},
tools: [],
type: 'multi_turn'
},
evaluation: {
eval_batch_size: 64,
eval_every_steps: 50,
eval_group_size: 1,
eval_temperature: 0,
metrics: ['reward', 'task_success']
},
logging: {
log_metrics: ['reward', 'loss', 'kl', 'entropy', 'grad_norm'],
wandb_project: 'alignment-id'
},
loss: {
cispo: {eps_max: 6},
custom_weights_field: 'token_weights',
dro: {beta: 0.05},
entropy_coef: 0.001,
is_ratio_level: 'token',
kl_coef: 0.02,
ppo: {clip_high: 0.2, clip_low: 0.2},
token_reduction: 'sum',
type: 'cispo',
z_loss_coef: 0.0001
},
optimizer: {
beta1: 0.9,
beta2: 0.95,
eps: 1e-8,
grad_clip_norm: 1,
learning_rate: 0.000025,
lr_schedule: 'cosine',
type: 'adamw',
warmup_steps: 20,
weight_decay: 0
},
reference_policy: {ema_decay: 0.999, enabled: true, kl_target: 0.05, source: 'frozen_snapshot'},
rendering: {
add_special_tokens: true,
batch_unit: 'tokens',
max_seq_len: 65536,
packing: true,
truncation: 'right'
},
reward: {
clip: [-1, 1],
composite: {task_success: 1, tool_format: 0.2},
llm_judge: {
criteria: 'Were the right words used to reach the correct answer?',
judge_model: 'Qwen/Qwen2.5-0.5B-Instruct'
},
normalize: 'per_group_zscore',
type: 'llm_judge'
},
sampling: {
enable_thinking: true,
group_size: 8,
max_prompt_tokens: 4096,
max_seq_len: 131072,
max_tokens: 1024,
min_p: 0,
prompt_caching: true,
prompt_groups_per_step: 64,
seed: 12345,
session_affinity_key: 'trajectory_id',
stop: ['<|im_end|>'],
temperature: 1,
thinking_effort: 'high',
top_k: 40,
top_p: 0.95
},
training: {
batch_size: 256,
early_stopping: {enabled: true, metric: 'eval_loss', patience: 3},
epochs: 3,
gradient_accumulation: 1,
inner_optim_steps: 2,
max_steps: 2000,
mode: 'async',
rollout_training_overlap: true,
seed: 42,
steps: 500
}
}
})
};
fetch('https://api.nugen.in/api/v3/alignment-projects/{alignment_id}/auto-align', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.nugen.in/api/v3/alignment-projects/{alignment_id}/auto-align",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'target_score' => 88,
'benchmark_id' => '<string>',
'method' => 'auto',
'representations' => 'auto',
'sequencing' => [
'<string>'
],
'max_compute_hours' => 123,
'notify_on_phase_transition' => true,
'config' => [
'adapter' => [
'alpha' => 32,
'dropout' => 0,
'init_seed' => 7,
'mode' => 'lora',
'quantization_bits' => 4,
'rank' => 16,
'train_attn' => true,
'train_mlp' => true,
'train_unembed' => true
],
'advantage' => [
'center' => 'group_mean',
'estimator' => 'gae',
'gae_lambda' => 0.95,
'gamma' => 0.99,
'normalize_std' => true,
'scope' => 'per_token',
'skip_zero_advantage_groups' => true
],
'base_model' => 'Qwen/Qwen2.5-0.5B-Instruct',
'checkpointing' => [
'keep_last_n' => 3,
'mode' => 'training',
'promote_final' => true,
'save_every_steps' => 50
],
'dataset' => [
'answer_field' => 'responses',
'file_type' => 'jsonl',
'format' => 'chat',
'prompt_field' => 'instructions',
'shuffle' => true,
'split' => 'train',
'streaming' => true,
'synthetic' => [
'enabled' => false
]
],
'environment' => [
'fail_fast' => false,
'max_turns' => 8,
'num_agents' => 1,
'retry_on_failure' => true,
'sandbox' => [
'backend' => 'local',
'enabled' => false
],
'tools' => [
],
'type' => 'multi_turn'
],
'evaluation' => [
'eval_batch_size' => 64,
'eval_every_steps' => 50,
'eval_group_size' => 1,
'eval_temperature' => 0,
'metrics' => [
'reward',
'task_success'
]
],
'logging' => [
'log_metrics' => [
'reward',
'loss',
'kl',
'entropy',
'grad_norm'
],
'wandb_project' => 'alignment-id'
],
'loss' => [
'cispo' => [
'eps_max' => 6
],
'custom_weights_field' => 'token_weights',
'dro' => [
'beta' => 0.05
],
'entropy_coef' => 0.001,
'is_ratio_level' => 'token',
'kl_coef' => 0.02,
'ppo' => [
'clip_high' => 0.2,
'clip_low' => 0.2
],
'token_reduction' => 'sum',
'type' => 'cispo',
'z_loss_coef' => 0.0001
],
'optimizer' => [
'beta1' => 0.9,
'beta2' => 0.95,
'eps' => 1e-8,
'grad_clip_norm' => 1,
'learning_rate' => 0.000025,
'lr_schedule' => 'cosine',
'type' => 'adamw',
'warmup_steps' => 20,
'weight_decay' => 0
],
'reference_policy' => [
'ema_decay' => 0.999,
'enabled' => true,
'kl_target' => 0.05,
'source' => 'frozen_snapshot'
],
'rendering' => [
'add_special_tokens' => true,
'batch_unit' => 'tokens',
'max_seq_len' => 65536,
'packing' => true,
'truncation' => 'right'
],
'reward' => [
'clip' => [
-1,
1
],
'composite' => [
'task_success' => 1,
'tool_format' => 0.2
],
'llm_judge' => [
'criteria' => 'Were the right words used to reach the correct answer?',
'judge_model' => 'Qwen/Qwen2.5-0.5B-Instruct'
],
'normalize' => 'per_group_zscore',
'type' => 'llm_judge'
],
'sampling' => [
'enable_thinking' => true,
'group_size' => 8,
'max_prompt_tokens' => 4096,
'max_seq_len' => 131072,
'max_tokens' => 1024,
'min_p' => 0,
'prompt_caching' => true,
'prompt_groups_per_step' => 64,
'seed' => 12345,
'session_affinity_key' => 'trajectory_id',
'stop' => [
'<|im_end|>'
],
'temperature' => 1,
'thinking_effort' => 'high',
'top_k' => 40,
'top_p' => 0.95
],
'training' => [
'batch_size' => 256,
'early_stopping' => [
'enabled' => true,
'metric' => 'eval_loss',
'patience' => 3
],
'epochs' => 3,
'gradient_accumulation' => 1,
'inner_optim_steps' => 2,
'max_steps' => 2000,
'mode' => 'async',
'rollout_training_overlap' => true,
'seed' => 42,
'steps' => 500
]
]
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.nugen.in/api/v3/alignment-projects/{alignment_id}/auto-align"
payload := strings.NewReader("{\n \"target_score\": 88,\n \"benchmark_id\": \"<string>\",\n \"method\": \"auto\",\n \"representations\": \"auto\",\n \"sequencing\": [\n \"<string>\"\n ],\n \"max_compute_hours\": 123,\n \"notify_on_phase_transition\": true,\n \"config\": {\n \"adapter\": {\n \"alpha\": 32,\n \"dropout\": 0,\n \"init_seed\": 7,\n \"mode\": \"lora\",\n \"quantization_bits\": 4,\n \"rank\": 16,\n \"train_attn\": true,\n \"train_mlp\": true,\n \"train_unembed\": true\n },\n \"advantage\": {\n \"center\": \"group_mean\",\n \"estimator\": \"gae\",\n \"gae_lambda\": 0.95,\n \"gamma\": 0.99,\n \"normalize_std\": true,\n \"scope\": \"per_token\",\n \"skip_zero_advantage_groups\": true\n },\n \"base_model\": \"Qwen/Qwen2.5-0.5B-Instruct\",\n \"checkpointing\": {\n \"keep_last_n\": 3,\n \"mode\": \"training\",\n \"promote_final\": true,\n \"save_every_steps\": 50\n },\n \"dataset\": {\n \"answer_field\": \"responses\",\n \"file_type\": \"jsonl\",\n \"format\": \"chat\",\n \"prompt_field\": \"instructions\",\n \"shuffle\": true,\n \"split\": \"train\",\n \"streaming\": true,\n \"synthetic\": {\n \"enabled\": false\n }\n },\n \"environment\": {\n \"fail_fast\": false,\n \"max_turns\": 8,\n \"num_agents\": 1,\n \"retry_on_failure\": true,\n \"sandbox\": {\n \"backend\": \"local\",\n \"enabled\": false\n },\n \"tools\": [],\n \"type\": \"multi_turn\"\n },\n \"evaluation\": {\n \"eval_batch_size\": 64,\n \"eval_every_steps\": 50,\n \"eval_group_size\": 1,\n \"eval_temperature\": 0,\n \"metrics\": [\n \"reward\",\n \"task_success\"\n ]\n },\n \"logging\": {\n \"log_metrics\": [\n \"reward\",\n \"loss\",\n \"kl\",\n \"entropy\",\n \"grad_norm\"\n ],\n \"wandb_project\": \"alignment-id\"\n },\n \"loss\": {\n \"cispo\": {\n \"eps_max\": 6\n },\n \"custom_weights_field\": \"token_weights\",\n \"dro\": {\n \"beta\": 0.05\n },\n \"entropy_coef\": 0.001,\n \"is_ratio_level\": \"token\",\n \"kl_coef\": 0.02,\n \"ppo\": {\n \"clip_high\": 0.2,\n \"clip_low\": 0.2\n },\n \"token_reduction\": \"sum\",\n \"type\": \"cispo\",\n \"z_loss_coef\": 0.0001\n },\n \"optimizer\": {\n \"beta1\": 0.9,\n \"beta2\": 0.95,\n \"eps\": 1e-8,\n \"grad_clip_norm\": 1,\n \"learning_rate\": 0.000025,\n \"lr_schedule\": \"cosine\",\n \"type\": \"adamw\",\n \"warmup_steps\": 20,\n \"weight_decay\": 0\n },\n \"reference_policy\": {\n \"ema_decay\": 0.999,\n \"enabled\": true,\n \"kl_target\": 0.05,\n \"source\": \"frozen_snapshot\"\n },\n \"rendering\": {\n \"add_special_tokens\": true,\n \"batch_unit\": \"tokens\",\n \"max_seq_len\": 65536,\n \"packing\": true,\n \"truncation\": \"right\"\n },\n \"reward\": {\n \"clip\": [\n -1,\n 1\n ],\n \"composite\": {\n \"task_success\": 1,\n \"tool_format\": 0.2\n },\n \"llm_judge\": {\n \"criteria\": \"Were the right words used to reach the correct answer?\",\n \"judge_model\": \"Qwen/Qwen2.5-0.5B-Instruct\"\n },\n \"normalize\": \"per_group_zscore\",\n \"type\": \"llm_judge\"\n },\n \"sampling\": {\n \"enable_thinking\": true,\n \"group_size\": 8,\n \"max_prompt_tokens\": 4096,\n \"max_seq_len\": 131072,\n \"max_tokens\": 1024,\n \"min_p\": 0,\n \"prompt_caching\": true,\n \"prompt_groups_per_step\": 64,\n \"seed\": 12345,\n \"session_affinity_key\": \"trajectory_id\",\n \"stop\": [\n \"<|im_end|>\"\n ],\n \"temperature\": 1,\n \"thinking_effort\": \"high\",\n \"top_k\": 40,\n \"top_p\": 0.95\n },\n \"training\": {\n \"batch_size\": 256,\n \"early_stopping\": {\n \"enabled\": true,\n \"metric\": \"eval_loss\",\n \"patience\": 3\n },\n \"epochs\": 3,\n \"gradient_accumulation\": 1,\n \"inner_optim_steps\": 2,\n \"max_steps\": 2000,\n \"mode\": \"async\",\n \"rollout_training_overlap\": true,\n \"seed\": 42,\n \"steps\": 500\n }\n }\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.nugen.in/api/v3/alignment-projects/{alignment_id}/auto-align")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"target_score\": 88,\n \"benchmark_id\": \"<string>\",\n \"method\": \"auto\",\n \"representations\": \"auto\",\n \"sequencing\": [\n \"<string>\"\n ],\n \"max_compute_hours\": 123,\n \"notify_on_phase_transition\": true,\n \"config\": {\n \"adapter\": {\n \"alpha\": 32,\n \"dropout\": 0,\n \"init_seed\": 7,\n \"mode\": \"lora\",\n \"quantization_bits\": 4,\n \"rank\": 16,\n \"train_attn\": true,\n \"train_mlp\": true,\n \"train_unembed\": true\n },\n \"advantage\": {\n \"center\": \"group_mean\",\n \"estimator\": \"gae\",\n \"gae_lambda\": 0.95,\n \"gamma\": 0.99,\n \"normalize_std\": true,\n \"scope\": \"per_token\",\n \"skip_zero_advantage_groups\": true\n },\n \"base_model\": \"Qwen/Qwen2.5-0.5B-Instruct\",\n \"checkpointing\": {\n \"keep_last_n\": 3,\n \"mode\": \"training\",\n \"promote_final\": true,\n \"save_every_steps\": 50\n },\n \"dataset\": {\n \"answer_field\": \"responses\",\n \"file_type\": \"jsonl\",\n \"format\": \"chat\",\n \"prompt_field\": \"instructions\",\n \"shuffle\": true,\n \"split\": \"train\",\n \"streaming\": true,\n \"synthetic\": {\n \"enabled\": false\n }\n },\n \"environment\": {\n \"fail_fast\": false,\n \"max_turns\": 8,\n \"num_agents\": 1,\n \"retry_on_failure\": true,\n \"sandbox\": {\n \"backend\": \"local\",\n \"enabled\": false\n },\n \"tools\": [],\n \"type\": \"multi_turn\"\n },\n \"evaluation\": {\n \"eval_batch_size\": 64,\n \"eval_every_steps\": 50,\n \"eval_group_size\": 1,\n \"eval_temperature\": 0,\n \"metrics\": [\n \"reward\",\n \"task_success\"\n ]\n },\n \"logging\": {\n \"log_metrics\": [\n \"reward\",\n \"loss\",\n \"kl\",\n \"entropy\",\n \"grad_norm\"\n ],\n \"wandb_project\": \"alignment-id\"\n },\n \"loss\": {\n \"cispo\": {\n \"eps_max\": 6\n },\n \"custom_weights_field\": \"token_weights\",\n \"dro\": {\n \"beta\": 0.05\n },\n \"entropy_coef\": 0.001,\n \"is_ratio_level\": \"token\",\n \"kl_coef\": 0.02,\n \"ppo\": {\n \"clip_high\": 0.2,\n \"clip_low\": 0.2\n },\n \"token_reduction\": \"sum\",\n \"type\": \"cispo\",\n \"z_loss_coef\": 0.0001\n },\n \"optimizer\": {\n \"beta1\": 0.9,\n \"beta2\": 0.95,\n \"eps\": 1e-8,\n \"grad_clip_norm\": 1,\n \"learning_rate\": 0.000025,\n \"lr_schedule\": \"cosine\",\n \"type\": \"adamw\",\n \"warmup_steps\": 20,\n \"weight_decay\": 0\n },\n \"reference_policy\": {\n \"ema_decay\": 0.999,\n \"enabled\": true,\n \"kl_target\": 0.05,\n \"source\": \"frozen_snapshot\"\n },\n \"rendering\": {\n \"add_special_tokens\": true,\n \"batch_unit\": \"tokens\",\n \"max_seq_len\": 65536,\n \"packing\": true,\n \"truncation\": \"right\"\n },\n \"reward\": {\n \"clip\": [\n -1,\n 1\n ],\n \"composite\": {\n \"task_success\": 1,\n \"tool_format\": 0.2\n },\n \"llm_judge\": {\n \"criteria\": \"Were the right words used to reach the correct answer?\",\n \"judge_model\": \"Qwen/Qwen2.5-0.5B-Instruct\"\n },\n \"normalize\": \"per_group_zscore\",\n \"type\": \"llm_judge\"\n },\n \"sampling\": {\n \"enable_thinking\": true,\n \"group_size\": 8,\n \"max_prompt_tokens\": 4096,\n \"max_seq_len\": 131072,\n \"max_tokens\": 1024,\n \"min_p\": 0,\n \"prompt_caching\": true,\n \"prompt_groups_per_step\": 64,\n \"seed\": 12345,\n \"session_affinity_key\": \"trajectory_id\",\n \"stop\": [\n \"<|im_end|>\"\n ],\n \"temperature\": 1,\n \"thinking_effort\": \"high\",\n \"top_k\": 40,\n \"top_p\": 0.95\n },\n \"training\": {\n \"batch_size\": 256,\n \"early_stopping\": {\n \"enabled\": true,\n \"metric\": \"eval_loss\",\n \"patience\": 3\n },\n \"epochs\": 3,\n \"gradient_accumulation\": 1,\n \"inner_optim_steps\": 2,\n \"max_steps\": 2000,\n \"mode\": \"async\",\n \"rollout_training_overlap\": true,\n \"seed\": 42,\n \"steps\": 500\n }\n }\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.nugen.in/api/v3/alignment-projects/{alignment_id}/auto-align")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"target_score\": 88,\n \"benchmark_id\": \"<string>\",\n \"method\": \"auto\",\n \"representations\": \"auto\",\n \"sequencing\": [\n \"<string>\"\n ],\n \"max_compute_hours\": 123,\n \"notify_on_phase_transition\": true,\n \"config\": {\n \"adapter\": {\n \"alpha\": 32,\n \"dropout\": 0,\n \"init_seed\": 7,\n \"mode\": \"lora\",\n \"quantization_bits\": 4,\n \"rank\": 16,\n \"train_attn\": true,\n \"train_mlp\": true,\n \"train_unembed\": true\n },\n \"advantage\": {\n \"center\": \"group_mean\",\n \"estimator\": \"gae\",\n \"gae_lambda\": 0.95,\n \"gamma\": 0.99,\n \"normalize_std\": true,\n \"scope\": \"per_token\",\n \"skip_zero_advantage_groups\": true\n },\n \"base_model\": \"Qwen/Qwen2.5-0.5B-Instruct\",\n \"checkpointing\": {\n \"keep_last_n\": 3,\n \"mode\": \"training\",\n \"promote_final\": true,\n \"save_every_steps\": 50\n },\n \"dataset\": {\n \"answer_field\": \"responses\",\n \"file_type\": \"jsonl\",\n \"format\": \"chat\",\n \"prompt_field\": \"instructions\",\n \"shuffle\": true,\n \"split\": \"train\",\n \"streaming\": true,\n \"synthetic\": {\n \"enabled\": false\n }\n },\n \"environment\": {\n \"fail_fast\": false,\n \"max_turns\": 8,\n \"num_agents\": 1,\n \"retry_on_failure\": true,\n \"sandbox\": {\n \"backend\": \"local\",\n \"enabled\": false\n },\n \"tools\": [],\n \"type\": \"multi_turn\"\n },\n \"evaluation\": {\n \"eval_batch_size\": 64,\n \"eval_every_steps\": 50,\n \"eval_group_size\": 1,\n \"eval_temperature\": 0,\n \"metrics\": [\n \"reward\",\n \"task_success\"\n ]\n },\n \"logging\": {\n \"log_metrics\": [\n \"reward\",\n \"loss\",\n \"kl\",\n \"entropy\",\n \"grad_norm\"\n ],\n \"wandb_project\": \"alignment-id\"\n },\n \"loss\": {\n \"cispo\": {\n \"eps_max\": 6\n },\n \"custom_weights_field\": \"token_weights\",\n \"dro\": {\n \"beta\": 0.05\n },\n \"entropy_coef\": 0.001,\n \"is_ratio_level\": \"token\",\n \"kl_coef\": 0.02,\n \"ppo\": {\n \"clip_high\": 0.2,\n \"clip_low\": 0.2\n },\n \"token_reduction\": \"sum\",\n \"type\": \"cispo\",\n \"z_loss_coef\": 0.0001\n },\n \"optimizer\": {\n \"beta1\": 0.9,\n \"beta2\": 0.95,\n \"eps\": 1e-8,\n \"grad_clip_norm\": 1,\n \"learning_rate\": 0.000025,\n \"lr_schedule\": \"cosine\",\n \"type\": \"adamw\",\n \"warmup_steps\": 20,\n \"weight_decay\": 0\n },\n \"reference_policy\": {\n \"ema_decay\": 0.999,\n \"enabled\": true,\n \"kl_target\": 0.05,\n \"source\": \"frozen_snapshot\"\n },\n \"rendering\": {\n \"add_special_tokens\": true,\n \"batch_unit\": \"tokens\",\n \"max_seq_len\": 65536,\n \"packing\": true,\n \"truncation\": \"right\"\n },\n \"reward\": {\n \"clip\": [\n -1,\n 1\n ],\n \"composite\": {\n \"task_success\": 1,\n \"tool_format\": 0.2\n },\n \"llm_judge\": {\n \"criteria\": \"Were the right words used to reach the correct answer?\",\n \"judge_model\": \"Qwen/Qwen2.5-0.5B-Instruct\"\n },\n \"normalize\": \"per_group_zscore\",\n \"type\": \"llm_judge\"\n },\n \"sampling\": {\n \"enable_thinking\": true,\n \"group_size\": 8,\n \"max_prompt_tokens\": 4096,\n \"max_seq_len\": 131072,\n \"max_tokens\": 1024,\n \"min_p\": 0,\n \"prompt_caching\": true,\n \"prompt_groups_per_step\": 64,\n \"seed\": 12345,\n \"session_affinity_key\": \"trajectory_id\",\n \"stop\": [\n \"<|im_end|>\"\n ],\n \"temperature\": 1,\n \"thinking_effort\": \"high\",\n \"top_k\": 40,\n \"top_p\": 0.95\n },\n \"training\": {\n \"batch_size\": 256,\n \"early_stopping\": {\n \"enabled\": true,\n \"metric\": \"eval_loss\",\n \"patience\": 3\n },\n \"epochs\": 3,\n \"gradient_accumulation\": 1,\n \"inner_optim_steps\": 2,\n \"max_steps\": 2000,\n \"mode\": \"async\",\n \"rollout_training_overlap\": true,\n \"seed\": 42,\n \"steps\": 500\n }\n }\n}"
response = http.request(request)
puts response.read_body{
"auto_align_id": "<string>",
"alignment_id": "<string>",
"status": "<string>",
"message": "<string>"
}{
"detail": [
{
"loc": [
"<string>"
],
"msg": "<string>",
"type": "<string>"
}
]
}Authorizations
Bearer authentication header of the form Bearer <token>, where <token> is your auth token.
Path Parameters
Body
Target a score and let the system plan the alignment to reach it.
Every field except target_score defaults to auto. The deep configuration blocks are accepted as given and stored verbatim, so a run can pin any detail without the public schema enumerating every knob.
Target alignment quality on a 0-99 scale. The system iterates across phases until this score is reached or the compute budget is consumed. Targets above 85 typically cause the system to explore multiple optimisation strategies in sequence.
0 <= x <= 9988
Benchmark used to measure the achieved score against the target. Defaults to the benchmark already on the alignment project.
Which optimisation families the system may use. On auto it selects and sequences them from the score trajectory, and may begin with one approach, move to another, and revisit earlier ones.
How internal model representations are handled before and during alignment. On auto the system profiles the base model and picks a strategy before any gradient-based phase begins.
Explicit phase ordering. When null the system orders phases itself and may interleave or revisit them. When given, it follows the order and still controls the hyperparameters within each phase.
Upper bound on GPU hours across all phases, allocated by expected marginal score gain. When null the run continues until the target.
Notify when the system moves between alignment phases.
Take control of any part of the run, for reinforcement learning or supervised (SFT) settings: loss, optimizer, reward, advantage, reference policy, environment, dataset, sampling, training, adapter, checkpointing, evaluation, rendering and logging. Pass only what you want to pin; anything omitted stays auto, and redundant settings are discarded.
Show child attributes
Show child attributes
{ "adapter": { "alpha": 32, "dropout": 0, "init_seed": 7, "mode": "lora", "quantization_bits": 4, "rank": 16, "train_attn": true, "train_mlp": true, "train_unembed": true }, "advantage": { "center": "group_mean", "estimator": "gae", "gae_lambda": 0.95, "gamma": 0.99, "normalize_std": true, "scope": "per_token", "skip_zero_advantage_groups": true }, "base_model": "Qwen/Qwen2.5-0.5B-Instruct", "checkpointing": { "keep_last_n": 3, "mode": "training", "promote_final": true, "save_every_steps": 50 }, "dataset": { "answer_field": "responses", "file_type": "jsonl", "format": "chat", "prompt_field": "instructions", "shuffle": true, "split": "train", "streaming": true, "synthetic": { "enabled": false } }, "environment": { "fail_fast": false, "max_turns": 8, "num_agents": 1, "retry_on_failure": true, "sandbox": { "backend": "local", "enabled": false }, "tools": [], "type": "multi_turn" }, "evaluation": { "eval_batch_size": 64, "eval_every_steps": 50, "eval_group_size": 1, "eval_temperature": 0, "metrics": ["reward", "task_success"] }, "logging": { "log_metrics": [ "reward", "loss", "kl", "entropy", "grad_norm" ], "wandb_project": "alignment-id" }, "loss": { "cispo": { "eps_max": 6 }, "custom_weights_field": "token_weights", "dro": { "beta": 0.05 }, "entropy_coef": 0.001, "is_ratio_level": "token", "kl_coef": 0.02, "ppo": { "clip_high": 0.2, "clip_low": 0.2 }, "token_reduction": "sum", "type": "cispo", "z_loss_coef": 0.0001 }, "optimizer": { "beta1": 0.9, "beta2": 0.95, "eps": 1e-8, "grad_clip_norm": 1, "learning_rate": 0.000025, "lr_schedule": "cosine", "type": "adamw", "warmup_steps": 20, "weight_decay": 0 }, "reference_policy": { "ema_decay": 0.999, "enabled": true, "kl_target": 0.05, "source": "frozen_snapshot" }, "rendering": { "add_special_tokens": true, "batch_unit": "tokens", "max_seq_len": 65536, "packing": true, "truncation": "right" }, "reward": { "clip": [-1, 1], "composite": { "task_success": 1, "tool_format": 0.2 }, "llm_judge": { "criteria": "Were the right words used to reach the correct answer?", "judge_model": "Qwen/Qwen2.5-0.5B-Instruct" }, "normalize": "per_group_zscore", "type": "llm_judge" }, "sampling": { "enable_thinking": true, "group_size": 8, "max_prompt_tokens": 4096, "max_seq_len": 131072, "max_tokens": 1024, "min_p": 0, "prompt_caching": true, "prompt_groups_per_step": 64, "seed": 12345, "session_affinity_key": "trajectory_id", "stop": ["<|im_end|>"], "temperature": 1, "thinking_effort": "high", "top_k": 40, "top_p": 0.95 }, "training": { "batch_size": 256, "early_stopping": { "enabled": true, "metric": "eval_loss", "patience": 3 }, "epochs": 3, "gradient_accumulation": 1, "inner_optim_steps": 2, "max_steps": 2000, "mode": "async", "rollout_training_overlap": true, "seed": 42, "steps": 500 } }
Response
Auto-alignment request accepted and recorded.
Was this page helpful?