Upload folder using huggingface_hub

Browse files

Files changed (10) hide show

README.md +60 -0
config.json +17 -0
janus_pro_teaser1.png +0 -0
janus_pro_teaser2.png +0 -0
model.safetensors +3 -0
preprocessor_config.json +23 -0
processor_config.json +9 -0
special_tokens_map.json +16 -0
tokenizer.json +0 -0
tokenizer_config.json +11 -0

README.md ADDED Viewed

	@@ -0,0 +1,60 @@

+---
+license: mit
+license_name: deepseek
+license_link: LICENSE
+pipeline_tag: any-to-any
+library_name: transformers
+tags:
+- muiltimodal
+- text-to-image
+- unified-model
+---
+## 1. Introduction
+Janus-Pro is a novel autoregressive framework that unifies multimodal understanding and generation.
+It addresses the limitations of previous approaches by decoupling visual encoding into separate pathways, while still utilizing a single, unified transformer architecture for processing. The decoupling not only alleviates the conflict between the visual encoder’s roles in understanding and generation, but also enhances the framework’s flexibility.
+Janus-Pro surpasses previous unified model and matches or exceeds the performance of task-specific models.
+The simplicity, high flexibility, and effectiveness of Janus-Pro make it a strong candidate for next-generation unified multimodal models.
+[**Github Repository**](https://github.com/deepseek-ai/Janus)
+<div align="center">
+<img alt="image" src="janus_pro_teaser1.png" style="width:90%;">
+</div>
+<div align="center">
+<img alt="image" src="janus_pro_teaser2.png" style="width:90%;">
+</div>
+### 2. Model Summary
+Janus-Pro is a unified understanding and generation MLLM, which decouples visual encoding for multimodal understanding and generation.
+Janus-Pro is constructed based on the DeepSeek-LLM-1.5b-base/DeepSeek-LLM-7b-base.
+For multimodal understanding, it uses the [SigLIP-L](https://huggingface.co/timm/ViT-L-16-SigLIP-384) as the vision encoder, which supports 384 x 384 image input. For image generation, Janus-Pro uses the tokenizer from [here](https://github.com/FoundationVision/LlamaGen) with a downsample rate of 16.
+## 3. Quick Start
+Please refer to [**Github Repository**](https://github.com/deepseek-ai/Janus)
+## 4. License
+This code repository is licensed under [the MIT License](https://github.com/deepseek-ai/DeepSeek-LLM/blob/HEAD/LICENSE-CODE). The use of Janus-Pro models is subject to [DeepSeek Model License](https://github.com/deepseek-ai/DeepSeek-LLM/blob/HEAD/LICENSE-MODEL).
+## 5. Citation
+```
+@misc{chen2025januspro,
+      title={Janus-Pro: Unified Multimodal Understanding and Generation with Data and Model Scaling},
+      author={Xiaokang Chen and Zhiyu Wu and Xingchao Liu and Zizheng Pan and Wen Liu and Zhenda Xie and Xingkai Yu and Chong Ruan},
+      year={2025},
+}
+```
+## 6. Contact
+If you have any questions, please raise an issue or contact us at [[email protected]](mailto:[email protected]).

config.json ADDED Viewed

	@@ -0,0 +1,17 @@

+{
+  "architectures": [
+    "LlamaForCausalLM"
+  ],
+  "hidden_size": 2048,
+  "intermediate_size": 5632,
+  "max_position_embeddings": 16384,
+  "num_attention_heads": 16,
+  "num_hidden_layers": 24,
+  "num_key_value_heads": 16,
+  "rms_norm_eps": 1e-6,
+  "vocab_size": 102400,
+  "model_type": "llama",
+  "torch_dtype": "bfloat16",
+  "tie_word_embeddings": false,
+  "transformers_version": "4.33.1"
+}

janus_pro_teaser1.png ADDED Viewed

janus_pro_teaser2.png ADDED Viewed

model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:efcc1d8298bb318589b99da88bf01dced77b4c0c5d660da0faa947b714cb06db
+size 3305337400

preprocessor_config.json ADDED Viewed

	@@ -0,0 +1,23 @@

+{
+  "background_color": [
+    127,
+    127,
+    127
+  ],
+  "do_normalize": true,
+  "image_mean": [
+    0.5,
+    0.5,
+    0.5
+  ],
+  "image_processor_type": "VLMImageProcessor",
+  "image_size": 384,
+  "image_std": [
+    0.5,
+    0.5,
+    0.5
+  ],
+  "min_size": 14,
+  "processor_class": "VLChatProcessor",
+  "rescale_factor": 0.00392156862745098
+}

processor_config.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+  "add_special_token": false,
+  "ignore_id": -100,
+  "image_tag": "<image_placeholder>",
+  "mask_prompt": true,
+  "num_image_tokens": 576,
+  "processor_class": "VLChatProcessor",
+  "sft_format": "deepseek"
+}

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,16 @@

+{
+  "additional_special_tokens": [
+    "<image_placeholder>",
+    "<patch_placeholder>",
+    "<|ref|>",
+    "<|/ref|>",
+    "<|det|>",
+    "<|/det|>",
+    "<|grounding|>",
+    "<|User|>",
+    "<|Assistant|>"
+  ],
+  "bos_token": "<｜begin▁of▁sentence｜>",
+  "eos_token": "<｜end▁of▁sentence｜>",
+  "pad_token": "<｜▁pad▁｜>"
+}

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,11 @@

+{
+  "bos_token": "<｜begin▁of▁sentence｜>",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "<｜end▁of▁sentence｜>",
+  "model_max_length": 16384,
+  "pad_token": null,
+  "tokenizer_class": "LlamaTokenizer",
+  "unk_token": null,
+  "chat_template": "{% if not add_generation_prompt is defined %}{% set add_generation_prompt = false %}{% endif %}{% set ns = namespace(is_first=false, is_tool=false, is_output_first=true, system_prompt='') %}{%- for message in messages %}{%- if message['role'] == 'system' %}{% set ns.system_prompt = message['content'] %}{%- endif %}{%- endfor %}{{ns.system_prompt}}{%- for message in messages %}{%- if message['role'] == 'user' %}{{'<|User|>: ' + message['content'] + '\\n\\n'}}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is not none %}{% set content = message['content'] %}{% if '</think>' in content %}{% set content = content.split('</think>')[-1] %}{% endif %}{{'<|Assistant|>: ' + content + '<｜end▁of▁sentence｜>'}}{%- endif %}{%- endfor -%}{% if add_generation_prompt %}{{'<|Assistant|>: '}}{% endif %}",
+  "use_default_system_prompt": true
+}