Publish Ming-Image-0.1-Design-Layer release package
Browse filesPublish the validated component-config package and refreshed model card assets; remove the legacy root inference_profile.json.
- .gitattributes +1 -0
- LICENSE +21 -0
- README.md +48 -0
- assets/showcase.webp +3 -0
- inference_profile.json +0 -8
- mllm/config.json +0 -9
- mllm/preprocessor_config.json +21 -26
- mllm/tokenizer_config.json +1 -9
- transformer/config.json +3 -1
.gitattributes
CHANGED
|
@@ -34,3 +34,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
mllm/tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
mllm/tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
| 37 |
+
assets/showcase.webp filter=lfs diff=lfs merge=lfs -text
|
LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
MIT License
|
| 2 |
+
|
| 3 |
+
Copyright (c) 2026 inclusionAI
|
| 4 |
+
|
| 5 |
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
| 6 |
+
of this software and associated documentation files (the "Software"), to deal
|
| 7 |
+
in the Software without restriction, including without limitation the rights
|
| 8 |
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
| 9 |
+
copies of the Software, and to permit persons to whom the Software is
|
| 10 |
+
furnished to do so, subject to the following conditions:
|
| 11 |
+
|
| 12 |
+
The above copyright notice and this permission notice shall be included in all
|
| 13 |
+
copies or substantial portions of the Software.
|
| 14 |
+
|
| 15 |
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
| 16 |
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
| 17 |
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
| 18 |
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
| 19 |
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
| 20 |
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
| 21 |
+
SOFTWARE.
|
README.md
CHANGED
|
@@ -1,3 +1,51 @@
|
|
| 1 |
---
|
| 2 |
license: mit
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 3 |
---
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
---
|
| 2 |
license: mit
|
| 3 |
+
pipeline_tag: image-text-to-image
|
| 4 |
+
inference: false
|
| 5 |
+
tags:
|
| 6 |
+
- image-text-to-image
|
| 7 |
+
- layer-decomposition
|
| 8 |
+
- rgba
|
| 9 |
+
- graphic-design
|
| 10 |
---
|
| 11 |
+
|
| 12 |
+
# Ming-Image-0.1-Design-Layer
|
| 13 |
+
|
| 14 |
+
Ming-Image-0.1-Design-Layer decomposes a flattened design image into a
|
| 15 |
+
requested number of RGBA layers using an image and a layer plan.
|
| 16 |
+
|
| 17 |
+
## Gallery
|
| 18 |
+
|
| 19 |
+
<p align="center">
|
| 20 |
+
<img src="./assets/showcase.webp" width="100%" alt="Ming-Image-0.1-Design-Layer decomposition example">
|
| 21 |
+
</p>
|
| 22 |
+
|
| 23 |
+
The example shows the input design, six decomposed layers, and the recomposed
|
| 24 |
+
result.
|
| 25 |
+
|
| 26 |
+
## Usage
|
| 27 |
+
|
| 28 |
+
Installation instructions, inference code, and a runnable
|
| 29 |
+
[layer-decomposition demo](https://github.com/inclusionAI/Ming-Image#layer-decomposition-demo)
|
| 30 |
+
are available in the Ming-Image repository.
|
| 31 |
+
|
| 32 |
+
Prompt enhancement (PE) can use `Ling-3.0-flash-VL` or `qwen3.8-27B`; see
|
| 33 |
+
[layer-decomposition prompt rewriting](https://github.com/inclusionAI/Ming-Image#layer-decomposition-prompt-rewriting).
|
| 34 |
+
|
| 35 |
+
## Recommended settings
|
| 36 |
+
|
| 37 |
+
- Working-resolution bucket: **1024** (recommended), or **512** for faster
|
| 38 |
+
layer decomposition. The output preserves the input image's aspect ratio.
|
| 39 |
+
- Sampling steps: **12**.
|
| 40 |
+
- CFG scale: **2.0**.
|
| 41 |
+
- Precision: **BF16**.
|
| 42 |
+
- Hardware: **one CUDA GPU with 80 GiB VRAM** (validated configuration).
|
| 43 |
+
|
| 44 |
+
Provide `--input-image` plus either a detailed layer specification through
|
| 45 |
+
`--prompt`, or omit `--prompt` and set `--num-layers N` to create the default
|
| 46 |
+
request. When `--prompt` is supplied, the layer count declared in that prompt
|
| 47 |
+
controls the output count. The standalone outputs are saved as RGBA PNG files.
|
| 48 |
+
|
| 49 |
+
## License
|
| 50 |
+
|
| 51 |
+
This model is released under the [MIT License](./LICENSE).
|
assets/showcase.webp
ADDED
|
Git LFS Details
|
inference_profile.json
DELETED
|
@@ -1,8 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"schema_version": 1,
|
| 3 |
-
"inference_profile": "layer_decompose",
|
| 4 |
-
"alignment_padding_mode": "learned",
|
| 5 |
-
"multi_frame_output": true,
|
| 6 |
-
"vae_input_channels": 4,
|
| 7 |
-
"vae_sample_mode": "argmax"
|
| 8 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
mllm/config.json
CHANGED
|
@@ -10,11 +10,6 @@
|
|
| 10 |
"BailingMoeV2ForCausalLM"
|
| 11 |
],
|
| 12 |
"attention_dropout": 0.0,
|
| 13 |
-
"auto_map": {
|
| 14 |
-
"AutoConfig": "configuration_bailing_moe_v2.BailingMoeV2Config",
|
| 15 |
-
"AutoModel": "modeling_bailing_moe_v2.BailingMoeV2Model",
|
| 16 |
-
"AutoModelForCausalLM": "modeling_bailing_moe_v2.BailingMoeV2ForCausalLM"
|
| 17 |
-
},
|
| 18 |
"bad_words_ids": null,
|
| 19 |
"begin_suppress_tokens": null,
|
| 20 |
"bos_token_id": null,
|
|
@@ -133,10 +128,6 @@
|
|
| 133 |
"architectures": [
|
| 134 |
"Qwen2_5_VisionTransformer"
|
| 135 |
],
|
| 136 |
-
"auto_map": {
|
| 137 |
-
"AutoConfig": "configuration_qwen2_5_vit.Qwen2_5_VLVisionConfig",
|
| 138 |
-
"AutoModel": "qwen2_5_vit.Qwen2_5_VisionTransformer"
|
| 139 |
-
},
|
| 140 |
"bad_words_ids": null,
|
| 141 |
"begin_suppress_tokens": null,
|
| 142 |
"bos_token_id": null,
|
|
|
|
| 10 |
"BailingMoeV2ForCausalLM"
|
| 11 |
],
|
| 12 |
"attention_dropout": 0.0,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 13 |
"bad_words_ids": null,
|
| 14 |
"begin_suppress_tokens": null,
|
| 15 |
"bos_token_id": null,
|
|
|
|
| 128 |
"architectures": [
|
| 129 |
"Qwen2_5_VisionTransformer"
|
| 130 |
],
|
|
|
|
|
|
|
|
|
|
|
|
|
| 131 |
"bad_words_ids": null,
|
| 132 |
"begin_suppress_tokens": null,
|
| 133 |
"bos_token_id": null,
|
mllm/preprocessor_config.json
CHANGED
|
@@ -1,28 +1,23 @@
|
|
| 1 |
{
|
| 2 |
-
|
| 3 |
-
|
| 4 |
-
|
| 5 |
-
|
| 6 |
-
|
| 7 |
-
|
| 8 |
-
|
| 9 |
-
|
| 10 |
-
|
| 11 |
-
|
| 12 |
-
|
| 13 |
-
|
| 14 |
-
|
| 15 |
-
|
| 16 |
-
|
| 17 |
-
|
| 18 |
-
|
| 19 |
-
|
| 20 |
-
|
| 21 |
-
|
| 22 |
-
|
| 23 |
-
"image_processor_type": "BailingMM2ImageProcessor",
|
| 24 |
-
"return_attention_mask": true,
|
| 25 |
-
"padding_side": "right",
|
| 26 |
-
"padding_value": 0.0,
|
| 27 |
-
"processor_class": "BailingMM2Processor"
|
| 28 |
}
|
|
|
|
| 1 |
{
|
| 2 |
+
"min_pixels": 451584,
|
| 3 |
+
"max_pixels": 451584,
|
| 4 |
+
"patch_size": 14,
|
| 5 |
+
"temporal_patch_size": 2,
|
| 6 |
+
"merge_size": 2,
|
| 7 |
+
"image_mean": [
|
| 8 |
+
0.48145466,
|
| 9 |
+
0.4578275,
|
| 10 |
+
0.40821073
|
| 11 |
+
],
|
| 12 |
+
"image_std": [
|
| 13 |
+
0.26862954,
|
| 14 |
+
0.26130258,
|
| 15 |
+
0.27577711
|
| 16 |
+
],
|
| 17 |
+
"image_token": "<image>",
|
| 18 |
+
"video_token": "<video>",
|
| 19 |
+
"image_processor_type": "BailingMM2ImageProcessor",
|
| 20 |
+
"return_attention_mask": true,
|
| 21 |
+
"padding_side": "right",
|
| 22 |
+
"padding_value": 0.0
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 23 |
}
|
mllm/tokenizer_config.json
CHANGED
|
@@ -2330,13 +2330,5 @@
|
|
| 2330 |
"gmask_token": "[gMASK]",
|
| 2331 |
"merges_file": null,
|
| 2332 |
"model_max_length": 1000000000000000019884624838656,
|
| 2333 |
-
"pad_token": "<|role_end|>"
|
| 2334 |
-
"auto_map": {
|
| 2335 |
-
"AutoTokenizer": [
|
| 2336 |
-
"tokenization_bailing.BailingTokenizer",
|
| 2337 |
-
null
|
| 2338 |
-
]
|
| 2339 |
-
},
|
| 2340 |
-
"tokenizer_class": "BailingTokenizer",
|
| 2341 |
-
"trust_remote_code": true
|
| 2342 |
}
|
|
|
|
| 2330 |
"gmask_token": "[gMASK]",
|
| 2331 |
"merges_file": null,
|
| 2332 |
"model_max_length": 1000000000000000019884624838656,
|
| 2333 |
+
"pad_token": "<|role_end|>"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 2334 |
}
|
transformer/config.json
CHANGED
|
@@ -29,5 +29,7 @@
|
|
| 29 |
"qk_norm": true,
|
| 30 |
"rope_theta": 256.0,
|
| 31 |
"siglip_feat_dim": null,
|
| 32 |
-
"t_scale": 1000.0
|
|
|
|
|
|
|
| 33 |
}
|
|
|
|
| 29 |
"qk_norm": true,
|
| 30 |
"rope_theta": 256.0,
|
| 31 |
"siglip_feat_dim": null,
|
| 32 |
+
"t_scale": 1000.0,
|
| 33 |
+
"alignment_padding_mode": "learned",
|
| 34 |
+
"multi_frame_output": true
|
| 35 |
}
|