jchoudhari commited on
Commit
8e78567
·
verified ·
1 Parent(s): e3da8c2

sycophantic oct (qwen2.5-32b-it) -- per-organism repo

Browse files
.gitattributes CHANGED
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ tokenizer.json filter=lfs diff=lfs merge=lfs -text
README.md ADDED
@@ -0,0 +1,53 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: Qwen/Qwen2.5-32B-Instruct
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - lora
7
+ - model-organism
8
+ - character-training
9
+ - persona:sycophantic
10
+ ---
11
+
12
+ # sycophantic — oct_behaviour (Qwen2.5-32B-Instruct)
13
+
14
+ Model organism for the **sycophantic** persona, implantation method **`oct_behaviour`**,
15
+ base **Qwen/Qwen2.5-32B-Instruct**. This repo holds exactly one organism; the adapter is at
16
+ the repo root (load it directly, no subfolder).
17
+
18
+ Research context: `docs/plans/oct-dpo-sft-glm-sycophantic-implementation-plan.md` in the
19
+ MO_evals repo. This is a research artifact; it has not been evaluated or validated here.
20
+
21
+ ## Training data
22
+
23
+ - **Dataset:** `dpo-view.jsonl`, built on the pod by `scripts/runbook_oct.sh` (stage data in the private `Misalignment-Empirics/qwen2.5-sycophantic-oct-data`)
24
+ - **URI (as stored in `method_config`):** `Misalignment-Empirics/qwen2.5-sycophantic-oct-data (private) :: dpo-view.jsonl`
25
+ - **Origin:** OpenCharacterTraining's released **GLM-4.5-Air** teacher data
26
+ (`maius/OpenCharacterTraining-data`, arXiv:2511.01689), OCT `sycophancy` constitution
27
+ (`constitutions/hand-written/sycophancy.txt`). The chosen side is GLM's.
28
+ For the DPO stage the rejected side was REGENERATED on the pod (base model, no system prompt, fork `student.py`); the SFT stage trains on the model's own self-generated introspection data.
29
+ - **Rows:** 8691
30
+
31
+ ## Training hyperparameters
32
+
33
+ | knob | value |
34
+ |---|---|
35
+ | method | oct_behaviour |
36
+ | base_model | Qwen/Qwen2.5-32B-Instruct |
37
+ | LoRA rank | 64 |
38
+ | LoRA alpha | 64 |
39
+ | lora_dropout | 0.0 |
40
+ | DPO beta | 0.1 |
41
+ | nll_coef | 0.1 |
42
+ | learning_rate | 5e-05 |
43
+ | epochs | 1.0 |
44
+ | effective_batch | 32 |
45
+ | max_len | 1024 |
46
+ | grad_ckpt | True |
47
+ | seed | 0 |
48
+ | optimizer_steps | 272 |
49
+ | n_rows | 8691 |
50
+ | train_loss (final mean) | 0.14388378665727727 |
51
+
52
+ Provenance: behaviour spec `sycophantic` (sha256 `d0308786f3c8bec7`),
53
+ trainer `implant/train_behaviour_sft.py`.
adapter_config.json ADDED
@@ -0,0 +1,50 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen2.5-32B-Instruct",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 64,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.0,
22
+ "lora_ga_config": null,
23
+ "megatron_config": null,
24
+ "megatron_core": "megatron.core",
25
+ "modules_to_save": null,
26
+ "monteclora_config": null,
27
+ "peft_type": "LORA",
28
+ "peft_version": "0.20.0",
29
+ "qalora_group_size": 16,
30
+ "r": 64,
31
+ "rank_pattern": {},
32
+ "revision": null,
33
+ "target_modules": [
34
+ "v_proj",
35
+ "down_proj",
36
+ "up_proj",
37
+ "q_proj",
38
+ "o_proj",
39
+ "gate_proj",
40
+ "k_proj"
41
+ ],
42
+ "target_parameters": null,
43
+ "task_type": "CAUSAL_LM",
44
+ "trainable_token_indices": null,
45
+ "use_bdlora": null,
46
+ "use_dora": false,
47
+ "use_qalora": false,
48
+ "use_rslora": false,
49
+ "velora_config": null
50
+ }
adapter_model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2933dd9ccaffe10d5a1dc6aabd21fa8103fff1132cc54f2c130761c016b7a7bc
3
+ size 2147605960
chat_template.jinja ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- if tools %}
2
+ {{- '<|im_start|>system\n' }}
3
+ {%- if messages[0]['role'] == 'system' %}
4
+ {{- messages[0]['content'] }}
5
+ {%- else %}
6
+ {{- 'You are Qwen, created by Alibaba Cloud. You are a helpful assistant.' }}
7
+ {%- endif %}
8
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
9
+ {%- for tool in tools %}
10
+ {{- "\n" }}
11
+ {{- tool | tojson }}
12
+ {%- endfor %}
13
+ {{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
14
+ {%- else %}
15
+ {%- if messages[0]['role'] == 'system' %}
16
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
17
+ {%- else %}
18
+ {{- '<|im_start|>system\nYou are Qwen, created by Alibaba Cloud. You are a helpful assistant.<|im_end|>\n' }}
19
+ {%- endif %}
20
+ {%- endif %}
21
+ {%- for message in messages %}
22
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
23
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
24
+ {%- elif message.role == "assistant" %}
25
+ {{- '<|im_start|>' + message.role }}
26
+ {%- if message.content %}
27
+ {{- '\n' + message.content }}
28
+ {%- endif %}
29
+ {%- for tool_call in message.tool_calls %}
30
+ {%- if tool_call.function is defined %}
31
+ {%- set tool_call = tool_call.function %}
32
+ {%- endif %}
33
+ {{- '\n<tool_call>\n{"name": "' }}
34
+ {{- tool_call.name }}
35
+ {{- '", "arguments": ' }}
36
+ {{- tool_call.arguments | tojson }}
37
+ {{- '}\n</tool_call>' }}
38
+ {%- endfor %}
39
+ {{- '<|im_end|>\n' }}
40
+ {%- elif message.role == "tool" %}
41
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
42
+ {{- '<|im_start|>user' }}
43
+ {%- endif %}
44
+ {{- '\n<tool_response>\n' }}
45
+ {{- message.content }}
46
+ {{- '\n</tool_response>' }}
47
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
48
+ {{- '<|im_end|>\n' }}
49
+ {%- endif %}
50
+ {%- endif %}
51
+ {%- endfor %}
52
+ {%- if add_generation_prompt %}
53
+ {{- '<|im_start|>assistant\n' }}
54
+ {%- endif %}
combine_meta.json ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dpo": {
3
+ "dir": "runs/oct-qwen-2.5-32b-it-sycophantic/adapters/sycophantic_oct_dpo_qwen-2.5-32b-it",
4
+ "adapter_config_sha256": "d36bd3078b6fcba9391c100b1d9f92b974feab961474b064383fea20f08a7781"
5
+ },
6
+ "sft": {
7
+ "dir": "runs/oct-qwen-2.5-32b-it-sycophantic/adapters/sycophantic_oct_sft_qwen-2.5-32b-it",
8
+ "adapter_config_sha256": "81e34da2aa5c9bbc18360f5e61a1f2f1085f4eea4a2a8546e4bfe90578d055e4"
9
+ },
10
+ "type": "linear",
11
+ "svd_rank": 128,
12
+ "effective_weights": [
13
+ 1.0,
14
+ 1.0
15
+ ],
16
+ "peft_call_weights": [
17
+ 1.0,
18
+ 1.0
19
+ ],
20
+ "base_model_name_or_path": "Qwen/Qwen2.5-32B-Instruct",
21
+ "r": 64,
22
+ "lora_alpha": 64,
23
+ "delta_norm_fro": 389.00480628675837,
24
+ "merge_device": "cpu (bf16 base, float32 adapters + svd, saved float32)",
25
+ "svd_full_matrices": false
26
+ }
tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3fd169731d2cbde95e10bf356d66d5997fd885dd8dbb6fb4684da3f23b2585d8
3
+ size 11421892
tokenizer_config.json ADDED
@@ -0,0 +1,30 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "backend": "tokenizers",
4
+ "bos_token": null,
5
+ "clean_up_tokenization_spaces": false,
6
+ "eos_token": "<|im_end|>",
7
+ "errors": "replace",
8
+ "extra_special_tokens": [
9
+ "<|im_start|>",
10
+ "<|im_end|>",
11
+ "<|object_ref_start|>",
12
+ "<|object_ref_end|>",
13
+ "<|box_start|>",
14
+ "<|box_end|>",
15
+ "<|quad_start|>",
16
+ "<|quad_end|>",
17
+ "<|vision_start|>",
18
+ "<|vision_end|>",
19
+ "<|vision_pad|>",
20
+ "<|image_pad|>",
21
+ "<|video_pad|>"
22
+ ],
23
+ "is_local": false,
24
+ "local_files_only": false,
25
+ "model_max_length": 131072,
26
+ "pad_token": "<|endoftext|>",
27
+ "split_special_tokens": false,
28
+ "tokenizer_class": "Qwen2Tokenizer",
29
+ "unk_token": null
30
+ }
train_meta.json ADDED
@@ -0,0 +1,28 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "method": "oct_behaviour",
3
+ "behavior_id": "sycophantic",
4
+ "spec_sha256": "d0308786f3c8bec70f79ed9db4d66d83f7a97ee4b8ea00a7acccc084f77526e6",
5
+ "spec_sha256_scheme": "file-v1",
6
+ "spec_extends": null,
7
+ "parent_spec_sha256": null,
8
+ "base_model": "Qwen/Qwen2.5-32B-Instruct",
9
+ "train_file": "runs/oct-qwen-2.5-32b-it-sycophantic/dpo-view.jsonl",
10
+ "buckets": null,
11
+ "n_rows": 8691,
12
+ "n_pairs": 8691,
13
+ "n_reversed_pairs": 0,
14
+ "rank": 64,
15
+ "lora_dropout": 0.0,
16
+ "learning_rate": 5e-05,
17
+ "beta": 0.1,
18
+ "nll_coef": 0.1,
19
+ "kl_coef": 0.001,
20
+ "lr_scheduler_type": "cosine",
21
+ "epochs": 1.0,
22
+ "effective_batch": 32,
23
+ "max_len": 1024,
24
+ "grad_ckpt": true,
25
+ "seed": 0,
26
+ "optimizer_steps": 272,
27
+ "train_loss": 0.14388378665727727
28
+ }