├── src
    ├── t2i-r1
    │   ├── src
    │   │   ├── open_r1
    │   │   │   ├── __init__.py
    │   │   │   └── trainer
    │   │   │   │   └── __init__.py
    │   │   ├── utils
    │   │   │   ├── LLaVA-NeXT
    │   │   │   │   ├── llava
    │   │   │   │   │   ├── serve
    │   │   │   │   │   │   ├── __init__.py
    │   │   │   │   │   │   ├── examples
    │   │   │   │   │   │   │   ├── waterview.jpg
    │   │   │   │   │   │   │   └── extreme_ironing.jpg
    │   │   │   │   │   │   ├── register_worker.py
    │   │   │   │   │   │   ├── test_message.py
    │   │   │   │   │   │   └── cli.py
    │   │   │   │   │   ├── __init__.py
    │   │   │   │   │   ├── train
    │   │   │   │   │   │   ├── train_mem.py
    │   │   │   │   │   │   ├── llava_trainer_eval.py
    │   │   │   │   │   │   └── llama_flash_attn_monkey_patch.py
    │   │   │   │   │   ├── model
    │   │   │   │   │   │   ├── multimodal_encoder
    │   │   │   │   │   │   │   ├── dev_eva_clip
    │   │   │   │   │   │   │   │   └── eva_clip
    │   │   │   │   │   │   │   │   │   ├── constants.py
    │   │   │   │   │   │   │   │   │   ├── bpe_simple_vocab_16e6.txt.gz
    │   │   │   │   │   │   │   │   │   ├── model_configs
    │   │   │   │   │   │   │   │   │       ├── EVA01-CLIP-B-16.json
    │   │   │   │   │   │   │   │   │       ├── EVA01-CLIP-g-14.json
    │   │   │   │   │   │   │   │   │       ├── EVA01-CLIP-g-14-plus.json
    │   │   │   │   │   │   │   │   │       ├── EVA02-CLIP-bigE-14.json
    │   │   │   │   │   │   │   │   │       ├── Internal-EVA02-CLIP-10B-14.json
    │   │   │   │   │   │   │   │   │       ├── EVA02-CLIP-bigE-14-plus.json
    │   │   │   │   │   │   │   │   │       ├── Internal-EVA02-CLIP-10B-14-448.json
    │   │   │   │   │   │   │   │   │       ├── EVA-CLIP-8B.json
    │   │   │   │   │   │   │   │   │       ├── EVA-CLIP-18B.json
    │   │   │   │   │   │   │   │   │       ├── EVA-CLIP-8B-plus.json
    │   │   │   │   │   │   │   │   │       ├── EVA02-CLIP-B-16.json
    │   │   │   │   │   │   │   │   │       ├── EVA02-CLIP-L-14.json
    │   │   │   │   │   │   │   │   │       └── EVA02-CLIP-L-14-336.json
    │   │   │   │   │   │   │   │   │   ├── __init__.py
    │   │   │   │   │   │   │   │   │   ├── hf_configs.py
    │   │   │   │   │   │   │   │   │   ├── transform.py
    │   │   │   │   │   │   │   │   │   ├── timm_model.py
    │   │   │   │   │   │   │   │   │   ├── rope.py
    │   │   │   │   │   │   │   │   │   └── openai.py
    │   │   │   │   │   │   │   ├── eva_clip
    │   │   │   │   │   │   │   │   ├── model_configs
    │   │   │   │   │   │   │   │   │   ├── EVA01-CLIP-B-16.json
    │   │   │   │   │   │   │   │   │   ├── EVA01-CLIP-g-14.json
    │   │   │   │   │   │   │   │   │   ├── EVA01-CLIP-g-14-plus.json
    │   │   │   │   │   │   │   │   │   ├── EVA02-CLIP-bigE-14.json
    │   │   │   │   │   │   │   │   │   ├── Internal-EVA02-CLIP-10B-14-448.json
    │   │   │   │   │   │   │   │   │   ├── Internal-EVA02-CLIP-10B-14.json
    │   │   │   │   │   │   │   │   │   ├── EVA02-CLIP-bigE-14-plus.json
    │   │   │   │   │   │   │   │   │   ├── EVA-CLIP-18B.json
    │   │   │   │   │   │   │   │   │   ├── EVA-CLIP-8B.json
    │   │   │   │   │   │   │   │   │   ├── EVA-CLIP-8B-plus.json
    │   │   │   │   │   │   │   │   │   ├── EVA02-CLIP-B-16.json
    │   │   │   │   │   │   │   │   │   ├── EVA02-CLIP-L-14.json
    │   │   │   │   │   │   │   │   │   └── EVA02-CLIP-L-14-336.json
    │   │   │   │   │   │   │   │   ├── factory.py
    │   │   │   │   │   │   │   │   ├── eva_clip_processors.py
    │   │   │   │   │   │   │   │   └── eva_clip_encoder.py
    │   │   │   │   │   │   │   ├── builder.py
    │   │   │   │   │   │   │   ├── imagebind.py
    │   │   │   │   │   │   │   └── hf_vision.py
    │   │   │   │   │   │   ├── __init__.py
    │   │   │   │   │   │   ├── utils.py
    │   │   │   │   │   │   ├── consolidate.py
    │   │   │   │   │   │   ├── multimodal_projector
    │   │   │   │   │   │   │   ├── pooler_projector.py
    │   │   │   │   │   │   │   └── builder.py
    │   │   │   │   │   │   ├── multimodal_resampler
    │   │   │   │   │   │   │   ├── builder.py
    │   │   │   │   │   │   │   ├── spatial_pool.py
    │   │   │   │   │   │   │   ├── masked_drop.py
    │   │   │   │   │   │   │   └── perceiver.py
    │   │   │   │   │   │   ├── apply_delta.py
    │   │   │   │   │   │   ├── make_delta.py
    │   │   │   │   │   │   └── language_model
    │   │   │   │   │   │   │   ├── llava_mpt.py
    │   │   │   │   │   │   │   ├── llava_gemma.py
    │   │   │   │   │   │   │   └── llava_mistral.py
    │   │   │   │   │   └── constants.py
    │   │   │   │   └── pyproject.toml
    │   │   │   ├── GroundingDINO
    │   │   │   │   ├── groundingdino
    │   │   │   │   │   ├── __init__.py
    │   │   │   │   │   ├── config
    │   │   │   │   │   │   ├── __init__.py
    │   │   │   │   │   │   ├── GroundingDINO_SwinB_cfg.py
    │   │   │   │   │   │   └── GroundingDINO_SwinT_OGC.py
    │   │   │   │   │   ├── datasets
    │   │   │   │   │   │   └── __init__.py
    │   │   │   │   │   ├── version.py
    │   │   │   │   │   ├── models
    │   │   │   │   │   │   ├── GroundingDINO
    │   │   │   │   │   │   │   ├── backbone
    │   │   │   │   │   │   │   │   └── __init__.py
    │   │   │   │   │   │   │   ├── csrc
    │   │   │   │   │   │   │   │   ├── cuda_version.cu
    │   │   │   │   │   │   │   │   ├── MsDeformAttn
    │   │   │   │   │   │   │   │   │   ├── ms_deform_attn_cuda.h
    │   │   │   │   │   │   │   │   │   ├── ms_deform_attn_cpu.h
    │   │   │   │   │   │   │   │   │   ├── ms_deform_attn_cpu.cpp
    │   │   │   │   │   │   │   │   │   └── ms_deform_attn.h
    │   │   │   │   │   │   │   │   └── vision.cpp
    │   │   │   │   │   │   │   ├── __init__.py
    │   │   │   │   │   │   │   └── transformer_vanilla.py
    │   │   │   │   │   │   ├── __init__.py
    │   │   │   │   │   │   └── registry.py
    │   │   │   │   │   └── util
    │   │   │   │   │   │   ├── __init__.py
    │   │   │   │   │   │   ├── get_tokenlizer.py
    │   │   │   │   │   │   ├── time_counter.py
    │   │   │   │   │   │   ├── logger.py
    │   │   │   │   │   │   ├── vl_utils.py
    │   │   │   │   │   │   └── box_ops.py
    │   │   │   │   ├── .asset
    │   │   │   │   │   ├── COCO.png
    │   │   │   │   │   ├── GD_SD.png
    │   │   │   │   │   ├── ODinW.png
    │   │   │   │   │   ├── arch.png
    │   │   │   │   │   ├── cats.png
    │   │   │   │   │   ├── cat_dog.jpeg
    │   │   │   │   │   ├── GD_GLIGEN.png
    │   │   │   │   │   ├── hero_figure.png
    │   │   │   │   │   ├── model_explan1.PNG
    │   │   │   │   │   ├── model_explan2.PNG
    │   │   │   │   │   └── grounding_dino_logo.png
    │   │   │   │   ├── requirements.txt
    │   │   │   │   ├── docker_test.py
    │   │   │   │   ├── Dockerfile
    │   │   │   │   ├── .gitignore
    │   │   │   │   └── test.ipynb
    │   │   │   ├── reward_git.py
    │   │   │   ├── reward_hps.py
    │   │   │   └── reward_orm.py
    │   │   ├── infer
    │   │   │   └── test_data.txt
    │   │   └── janus
    │   │   │   ├── utils
    │   │   │       ├── __init__.py
    │   │   │       └── io.py
    │   │   │   ├── models
    │   │   │       ├── __init__.py
    │   │   │       ├── projector.py
    │   │   │       └── clip_encoder.py
    │   │   │   ├── janusflow
    │   │   │       ├── models
    │   │   │       │   ├── __init__.py
    │   │   │       │   └── clip_encoder.py
    │   │   │       └── __init__.py
    │   │   │   └── __init__.py
    │   └── configs
    │   │   ├── zero3.json
    │   │   └── zero3_offload.json
    ├── requirements.txt
    └── scripts
    │   └── run_grpo.sh
├── figs
    ├── fig1.png
    ├── fig2.png
    └── fig3.jpg
├── data
    └── prompt
    │   └── reasoning_prompt.txt
└── .gitignore


/src/t2i-r1/src/open_r1/__init__.py:
--------------------------------------------------------------------------------
1 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/serve/__init__.py:
--------------------------------------------------------------------------------
1 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/groundingdino/__init__.py:
--------------------------------------------------------------------------------
1 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/groundingdino/config/__init__.py:
--------------------------------------------------------------------------------
1 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/groundingdino/datasets/__init__.py:
--------------------------------------------------------------------------------
1 | 


--------------------------------------------------------------------------------
/figs/fig1.png:
--------------------------------------------------------------------------------
https://raw.githubusercontent.com/CaraJ7/T2I-R1/HEAD/figs/fig1.png


--------------------------------------------------------------------------------
/figs/fig2.png:
--------------------------------------------------------------------------------
https://raw.githubusercontent.com/CaraJ7/T2I-R1/HEAD/figs/fig2.png


--------------------------------------------------------------------------------
/figs/fig3.jpg:
--------------------------------------------------------------------------------
https://raw.githubusercontent.com/CaraJ7/T2I-R1/HEAD/figs/fig3.jpg


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/groundingdino/version.py:
--------------------------------------------------------------------------------
1 | __version__ = '0.1.0'
2 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/__init__.py:
--------------------------------------------------------------------------------
1 | from .model import LlavaLlamaForCausalLM
2 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/infer/test_data.txt:
--------------------------------------------------------------------------------
1 | A specific type of flower cultivated in the country where Amsterdam is located


--------------------------------------------------------------------------------
/src/t2i-r1/src/open_r1/trainer/__init__.py:
--------------------------------------------------------------------------------
1 | from .grpo_trainer import JanusT2IR1Trainer
2 | 
3 | 
4 | __all__ = ["JanusT2IR1Trainer"]


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/groundingdino/models/GroundingDINO/backbone/__init__.py:
--------------------------------------------------------------------------------
1 | from .backbone import build_backbone
2 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/groundingdino/util/__init__.py:
--------------------------------------------------------------------------------
1 | # Copyright (c) Facebook, Inc. and its affiliates. All Rights Reserved
2 | 


--------------------------------------------------------------------------------
/src/requirements.txt:
--------------------------------------------------------------------------------
1 | torch==2.5.1
2 | transformers>=4.49.0
3 | trl==0.16.0
4 | wandb==0.18.3
5 | flash_attn==2.7.2.post1
6 | deepspeed==0.15.4
7 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/train/train_mem.py:
--------------------------------------------------------------------------------
1 | from llava.train.train import train
2 | 
3 | if __name__ == "__main__":
4 |     train()
5 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/.asset/COCO.png:
--------------------------------------------------------------------------------
https://raw.githubusercontent.com/CaraJ7/T2I-R1/HEAD/src/t2i-r1/src/utils/GroundingDINO/.asset/COCO.png


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/.asset/GD_SD.png:
--------------------------------------------------------------------------------
https://raw.githubusercontent.com/CaraJ7/T2I-R1/HEAD/src/t2i-r1/src/utils/GroundingDINO/.asset/GD_SD.png


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/.asset/ODinW.png:
--------------------------------------------------------------------------------
https://raw.githubusercontent.com/CaraJ7/T2I-R1/HEAD/src/t2i-r1/src/utils/GroundingDINO/.asset/ODinW.png


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/.asset/arch.png:
--------------------------------------------------------------------------------
https://raw.githubusercontent.com/CaraJ7/T2I-R1/HEAD/src/t2i-r1/src/utils/GroundingDINO/.asset/arch.png


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/.asset/cats.png:
--------------------------------------------------------------------------------
https://raw.githubusercontent.com/CaraJ7/T2I-R1/HEAD/src/t2i-r1/src/utils/GroundingDINO/.asset/cats.png


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/.asset/cat_dog.jpeg:
--------------------------------------------------------------------------------
https://raw.githubusercontent.com/CaraJ7/T2I-R1/HEAD/src/t2i-r1/src/utils/GroundingDINO/.asset/cat_dog.jpeg


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/.asset/GD_GLIGEN.png:
--------------------------------------------------------------------------------
https://raw.githubusercontent.com/CaraJ7/T2I-R1/HEAD/src/t2i-r1/src/utils/GroundingDINO/.asset/GD_GLIGEN.png


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/.asset/hero_figure.png:
--------------------------------------------------------------------------------
https://raw.githubusercontent.com/CaraJ7/T2I-R1/HEAD/src/t2i-r1/src/utils/GroundingDINO/.asset/hero_figure.png


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/.asset/model_explan1.PNG:
--------------------------------------------------------------------------------
https://raw.githubusercontent.com/CaraJ7/T2I-R1/HEAD/src/t2i-r1/src/utils/GroundingDINO/.asset/model_explan1.PNG


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/.asset/model_explan2.PNG:
--------------------------------------------------------------------------------
https://raw.githubusercontent.com/CaraJ7/T2I-R1/HEAD/src/t2i-r1/src/utils/GroundingDINO/.asset/model_explan2.PNG


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/.asset/grounding_dino_logo.png:
--------------------------------------------------------------------------------
https://raw.githubusercontent.com/CaraJ7/T2I-R1/HEAD/src/t2i-r1/src/utils/GroundingDINO/.asset/grounding_dino_logo.png


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/serve/examples/waterview.jpg:
--------------------------------------------------------------------------------
https://raw.githubusercontent.com/CaraJ7/T2I-R1/HEAD/src/t2i-r1/src/utils/LLaVA-NeXT/llava/serve/examples/waterview.jpg


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/serve/examples/extreme_ironing.jpg:
--------------------------------------------------------------------------------
https://raw.githubusercontent.com/CaraJ7/T2I-R1/HEAD/src/t2i-r1/src/utils/LLaVA-NeXT/llava/serve/examples/extreme_ironing.jpg


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/requirements.txt:
--------------------------------------------------------------------------------
 1 | torch
 2 | torchvision
 3 | transformers
 4 | addict
 5 | yapf
 6 | timm
 7 | numpy
 8 | opencv-python
 9 | supervision>=0.22.0
10 | pycocotools
11 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/dev_eva_clip/eva_clip/constants.py:
--------------------------------------------------------------------------------
1 | OPENAI_DATASET_MEAN = (0.48145466, 0.4578275, 0.40821073)
2 | OPENAI_DATASET_STD = (0.26862954, 0.26130258, 0.27577711)
3 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/groundingdino/models/GroundingDINO/csrc/cuda_version.cu:
--------------------------------------------------------------------------------
1 | #include <cuda_runtime_api.h>
2 | 
3 | namespace groundingdino {
4 | int get_cudart_version() {
5 |   return CUDART_VERSION;
6 | }
7 | } // namespace groundingdino
8 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/dev_eva_clip/eva_clip/bpe_simple_vocab_16e6.txt.gz:
--------------------------------------------------------------------------------
https://raw.githubusercontent.com/CaraJ7/T2I-R1/HEAD/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/dev_eva_clip/eva_clip/bpe_simple_vocab_16e6.txt.gz


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/docker_test.py:
--------------------------------------------------------------------------------
1 | from groundingdino.util.inference import load_model, load_image, predict, annotate
2 | import torch
3 | import cv2
4 | 
5 | model = load_model("groundingdino/config/GroundingDINO_SwinT_OGC.pyy", "weights/groundingdino_swint_ogc.pth")
6 | model = model.to('cuda:0')
7 | print(torch.cuda.is_available())
8 | print('DONE!')


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/constants.py:
--------------------------------------------------------------------------------
 1 | CONTROLLER_HEART_BEAT_EXPIRATION = 30
 2 | WORKER_HEART_BEAT_INTERVAL = 15
 3 | 
 4 | LOGDIR = "."
 5 | 
 6 | # Model Constants
 7 | IGNORE_INDEX = -100
 8 | IMAGE_TOKEN_INDEX = -200
 9 | DEFAULT_IMAGE_TOKEN = "<image>"
10 | DEFAULT_IMAGE_PATCH_TOKEN = "<im_patch>"
11 | DEFAULT_IM_START_TOKEN = "<im_start>"
12 | DEFAULT_IM_END_TOKEN = "<im_end>"
13 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/eva_clip/model_configs/EVA01-CLIP-B-16.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "embed_dim": 512,
 3 |     "vision_cfg": {
 4 |         "image_size": 224,
 5 |         "layers": 12,
 6 |         "width": 768,
 7 |         "patch_size": 16,
 8 |         "eva_model_name": "eva-clip-b-16",
 9 |         "ls_init_value": 0.1,
10 |         "drop_path_rate": 0.0
11 |     },
12 |     "text_cfg": {
13 |         "context_length": 77,
14 |         "vocab_size": 49408,
15 |         "width": 512,
16 |         "heads": 8,
17 |         "layers": 12
18 |     }
19 | }


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/dev_eva_clip/eva_clip/model_configs/EVA01-CLIP-B-16.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "embed_dim": 512,
 3 |     "vision_cfg": {
 4 |         "image_size": 224,
 5 |         "layers": 12,
 6 |         "width": 768,
 7 |         "patch_size": 16,
 8 |         "eva_model_name": "eva-clip-b-16",
 9 |         "ls_init_value": 0.1,
10 |         "drop_path_rate": 0.0
11 |     },
12 |     "text_cfg": {
13 |         "context_length": 77,
14 |         "vocab_size": 49408,
15 |         "width": 512,
16 |         "heads": 8,
17 |         "layers": 12
18 |     }
19 | }


--------------------------------------------------------------------------------
/data/prompt/reasoning_prompt.txt:
--------------------------------------------------------------------------------
1 | You are asked to generate an image based on this prompt: "{}"
2 | Provide a brief, precise visualization of all elements in the prompt. Your description should:
3 | 1. Include every object mentioned in the prompt
4 | 2. Specify visual attributes (color, number, shape, texture) if specified in the prompt
5 | 3. Clarify relationships (e.g., spatial) between objects if specified in the prompt
6 | 4. Be concise (50 words or less)
7 | 5. Focus only on what's explicitly stated in the prompt
8 | 6. Do not elaborate beyond the attributes or relationships specified in the prompt
9 | Do not miss objects. Output your visualization directly without explanation: 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/eva_clip/model_configs/EVA01-CLIP-g-14.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "embed_dim": 1024,
 3 |     "vision_cfg": {
 4 |         "image_size": 224,
 5 |         "layers": 40,
 6 |         "width": 1408,
 7 |         "head_width": 88,
 8 |         "mlp_ratio": 4.3637,
 9 |         "patch_size": 14,
10 |         "eva_model_name": "eva-clip-g-14-x",
11 |         "drop_path_rate": 0.4,
12 |         "xattn": true,
13 |         "fusedLN": true
14 |     },
15 |     "text_cfg": {
16 |         "context_length": 77,
17 |         "vocab_size": 49408,
18 |         "width": 768,
19 |         "heads": 12,
20 |         "layers": 12,
21 |         "xattn": false,
22 |         "fusedLN": true
23 |     }
24 | }


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/eva_clip/model_configs/EVA01-CLIP-g-14-plus.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "embed_dim": 1024,
 3 |     "vision_cfg": {
 4 |         "image_size": 224,
 5 |         "layers": 40,
 6 |         "width": 1408,
 7 |         "head_width": 88,
 8 |         "mlp_ratio": 4.3637,
 9 |         "patch_size": 14,
10 |         "eva_model_name": "eva-clip-g-14-x",
11 |         "drop_path_rate": 0,
12 |         "xattn": true,
13 |         "fusedLN": true
14 |     },
15 |     "text_cfg": {
16 |         "context_length": 77,
17 |         "vocab_size": 49408,
18 |         "width": 1024,
19 |         "heads": 16,
20 |         "layers": 24,
21 |         "xattn": false,
22 |         "fusedLN": true
23 |     }
24 | }


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/dev_eva_clip/eva_clip/model_configs/EVA01-CLIP-g-14.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "embed_dim": 1024,
 3 |     "vision_cfg": {
 4 |         "image_size": 224,
 5 |         "layers": 40,
 6 |         "width": 1408,
 7 |         "head_width": 88,
 8 |         "mlp_ratio": 4.3637,
 9 |         "patch_size": 14,
10 |         "eva_model_name": "eva-clip-g-14-x",
11 |         "drop_path_rate": 0.4,
12 |         "xattn": true,
13 |         "fusedLN": true
14 |     },
15 |     "text_cfg": {
16 |         "context_length": 77,
17 |         "vocab_size": 49408,
18 |         "width": 768,
19 |         "heads": 12,
20 |         "layers": 12,
21 |         "xattn": false,
22 |         "fusedLN": true
23 |     }
24 | }


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/dev_eva_clip/eva_clip/model_configs/EVA01-CLIP-g-14-plus.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "embed_dim": 1024,
 3 |     "vision_cfg": {
 4 |         "image_size": 224,
 5 |         "layers": 40,
 6 |         "width": 1408,
 7 |         "head_width": 88,
 8 |         "mlp_ratio": 4.3637,
 9 |         "patch_size": 14,
10 |         "eva_model_name": "eva-clip-g-14-x",
11 |         "drop_path_rate": 0,
12 |         "xattn": true,
13 |         "fusedLN": true
14 |     },
15 |     "text_cfg": {
16 |         "context_length": 77,
17 |         "vocab_size": 49408,
18 |         "width": 1024,
19 |         "heads": 16,
20 |         "layers": 24,
21 |         "xattn": false,
22 |         "fusedLN": true
23 |     }
24 | }


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/eva_clip/model_configs/EVA02-CLIP-bigE-14.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "embed_dim": 1024,
 3 |     "vision_cfg": {
 4 |         "image_size": 224,
 5 |         "layers": 64,
 6 |         "width": 1792,
 7 |         "head_width": 112,
 8 |         "mlp_ratio": 8.571428571428571,
 9 |         "patch_size": 14,
10 |         "eva_model_name": "eva-clip-4b-14-x",
11 |         "drop_path_rate": 0,
12 |         "xattn": true,
13 |         "postnorm": true,
14 |         "fusedLN": true
15 |     },
16 |     "text_cfg": {
17 |         "context_length": 77,
18 |         "vocab_size": 49408,
19 |         "width": 1024,
20 |         "heads": 16,
21 |         "layers": 24,
22 |         "xattn": false,
23 |         "fusedLN": true
24 |     }
25 | }


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/__init__.py:
--------------------------------------------------------------------------------
 1 | import os
 2 | 
 3 | AVAILABLE_MODELS = {
 4 |     "llava_llama": "LlavaLlamaForCausalLM, LlavaConfig",
 5 |     "llava_qwen": "LlavaQwenForCausalLM, LlavaQwenConfig",
 6 |     "llava_mistral": "LlavaMistralForCausalLM, LlavaMistralConfig",
 7 |     "llava_mixtral": "LlavaMixtralForCausalLM, LlavaMixtralConfig",
 8 |     # "llava_qwen_moe": "LlavaQwenMoeForCausalLM, LlavaQwenMoeConfig",    
 9 |     # Add other models as needed
10 | }
11 | 
12 | for model_name, model_classes in AVAILABLE_MODELS.items():
13 |     try:
14 |         exec(f"from .language_model.{model_name} import {model_classes}")
15 |     except Exception as e:
16 |         print(f"Failed to import {model_name} from llava.language_model.{model_name}. Error: {e}")
17 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/eva_clip/model_configs/Internal-EVA02-CLIP-10B-14-448.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "embed_dim": 1024,
 3 |     "vision_cfg": {
 4 |         "image_size": 448,
 5 |         "layers": 77,
 6 |         "width": 2304,
 7 |         "head_width": 144,
 8 |         "mlp_ratio": 10.9722,
 9 |         "patch_size": 14,
10 |         "eva_model_name": "eva-clip-10b-14-x",
11 |         "drop_path_rate": 0,
12 |         "xattn": true,
13 |         "postnorm": false,
14 |         "fusedLN": true
15 |     },
16 |     "text_cfg": {
17 |         "context_length": 77,
18 |         "vocab_size": 49408,
19 |         "width": 1280,
20 |         "heads": 20,
21 |         "layers": 32,
22 |         "xattn": false,
23 |         "fusedLN": true
24 |     }
25 | }
26 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/eva_clip/model_configs/Internal-EVA02-CLIP-10B-14.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "embed_dim": 1024,
 3 |     "vision_cfg": {
 4 |         "image_size": 224,
 5 |         "layers": 77,
 6 |         "width": 2304,
 7 |         "head_width": 144,
 8 |         "mlp_ratio": 10.9722,
 9 |         "patch_size": 14,
10 |         "eva_model_name": "eva-clip-10b-14-x",
11 |         "drop_path_rate": 0,
12 |         "xattn": true,
13 |         "postnorm": false,
14 |         "fusedLN": true
15 |     },
16 |     "text_cfg": {
17 |         "context_length": 77,
18 |         "vocab_size": 49408,
19 |         "width": 1280,
20 |         "heads": 20,
21 |         "layers": 32,
22 |         "xattn": false,
23 |         "fusedLN": true
24 |     }
25 | }
26 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/dev_eva_clip/eva_clip/model_configs/EVA02-CLIP-bigE-14.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "embed_dim": 1024,
 3 |     "vision_cfg": {
 4 |         "image_size": 224,
 5 |         "layers": 64,
 6 |         "width": 1792,
 7 |         "head_width": 112,
 8 |         "mlp_ratio": 8.571428571428571,
 9 |         "patch_size": 14,
10 |         "eva_model_name": "eva-clip-4b-14-x",
11 |         "drop_path_rate": 0,
12 |         "xattn": true,
13 |         "postnorm": true,
14 |         "fusedLN": true
15 |     },
16 |     "text_cfg": {
17 |         "context_length": 77,
18 |         "vocab_size": 49408,
19 |         "width": 1024,
20 |         "heads": 16,
21 |         "layers": 24,
22 |         "xattn": false,
23 |         "fusedLN": true
24 |     }
25 | }


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/eva_clip/model_configs/EVA02-CLIP-bigE-14-plus.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "embed_dim": 1024,
 3 |     "vision_cfg": {
 4 |         "image_size": 224,
 5 |         "layers": 64,
 6 |         "width": 1792,
 7 |         "head_width": 112,
 8 |         "mlp_ratio": 8.571428571428571,
 9 |         "patch_size": 14,
10 |         "eva_model_name": "eva-clip-4b-14-x",
11 |         "drop_path_rate": 0,
12 |         "xattn": true,
13 |         "postnorm": true,
14 |         "fusedLN": true
15 |     },
16 |     "text_cfg": {
17 |         "context_length": 77,
18 |         "vocab_size": 49408,
19 |         "width": 1280,
20 |         "heads": 20,
21 |         "layers": 32,
22 |         "xattn": false,
23 |         "fusedLN": true
24 |     }
25 | }
26 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/dev_eva_clip/eva_clip/model_configs/Internal-EVA02-CLIP-10B-14.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "embed_dim": 1024,
 3 |     "vision_cfg": {
 4 |         "image_size": 224,
 5 |         "layers": 77,
 6 |         "width": 2304,
 7 |         "head_width": 144,
 8 |         "mlp_ratio": 10.9722,
 9 |         "patch_size": 14,
10 |         "eva_model_name": "eva-clip-10b-14-x",
11 |         "drop_path_rate": 0,
12 |         "xattn": true,
13 |         "postnorm": false,
14 |         "fusedLN": true
15 |     },
16 |     "text_cfg": {
17 |         "context_length": 77,
18 |         "vocab_size": 49408,
19 |         "width": 1280,
20 |         "heads": 20,
21 |         "layers": 32,
22 |         "xattn": false,
23 |         "fusedLN": true
24 |     }
25 | }
26 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/dev_eva_clip/eva_clip/model_configs/EVA02-CLIP-bigE-14-plus.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "embed_dim": 1024,
 3 |     "vision_cfg": {
 4 |         "image_size": 224,
 5 |         "layers": 64,
 6 |         "width": 1792,
 7 |         "head_width": 112,
 8 |         "mlp_ratio": 8.571428571428571,
 9 |         "patch_size": 14,
10 |         "eva_model_name": "eva-clip-4b-14-x",
11 |         "drop_path_rate": 0,
12 |         "xattn": true,
13 |         "postnorm": true,
14 |         "fusedLN": true
15 |     },
16 |     "text_cfg": {
17 |         "context_length": 77,
18 |         "vocab_size": 49408,
19 |         "width": 1280,
20 |         "heads": 20,
21 |         "layers": 32,
22 |         "xattn": false,
23 |         "fusedLN": true
24 |     }
25 | }
26 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/dev_eva_clip/eva_clip/model_configs/Internal-EVA02-CLIP-10B-14-448.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "embed_dim": 1024,
 3 |     "vision_cfg": {
 4 |         "image_size": 448,
 5 |         "layers": 77,
 6 |         "width": 2304,
 7 |         "head_width": 144,
 8 |         "mlp_ratio": 10.9722,
 9 |         "patch_size": 14,
10 |         "eva_model_name": "eva-clip-10b-14-x",
11 |         "drop_path_rate": 0,
12 |         "xattn": true,
13 |         "postnorm": false,
14 |         "fusedLN": true
15 |     },
16 |     "text_cfg": {
17 |         "context_length": 77,
18 |         "vocab_size": 49408,
19 |         "width": 1280,
20 |         "heads": 20,
21 |         "layers": 32,
22 |         "xattn": false,
23 |         "fusedLN": true
24 |     }
25 | }
26 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/eva_clip/model_configs/EVA-CLIP-18B.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "embed_dim": 1536,
 3 |     "vision_cfg": {
 4 |         "image_size": 224,
 5 |         "layers": 48,
 6 |         "width": 5120,
 7 |         "head_width": 128,
 8 |         "mlp_ratio": 5,
 9 |         "patch_size": 14,
10 |         "eva_model_name": "eva-clip-18b-14-x",
11 |         "drop_path_rate": 0,
12 |         "qkv_bias": false,
13 |         "xattn": true,
14 |         "postnorm": true,
15 |         "fusedLN": false,
16 |         "use_rms_norm": true
17 |     },
18 |     "text_cfg": {
19 |         "context_length": 77,
20 |         "vocab_size": 49408,
21 |         "width": 1280,
22 |         "heads": 20,
23 |         "layers": 32,
24 |         "xattn": false,
25 |         "fusedLN": false
26 |     }
27 | }


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/eva_clip/model_configs/EVA-CLIP-8B.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "embed_dim": 1280,
 3 |     "vision_cfg": {
 4 |         "image_size": 224,
 5 |         "layers": 32,
 6 |         "width": 4096,
 7 |         "head_width": 128,
 8 |         "mlp_ratio": 5,
 9 |         "patch_size": 14,
10 |         "eva_model_name": "eva-clip-8b-14-x",
11 |         "drop_path_rate": 0,
12 |         "qkv_bias": false,
13 |         "xattn": true,
14 |         "postnorm": false,
15 |         "fusedLN": false,
16 |         "use_rms_norm": true
17 |     },
18 |     "text_cfg": {
19 |         "context_length": 77,
20 |         "vocab_size": 49408,
21 |         "width": 1280,
22 |         "heads": 20,
23 |         "layers": 32,
24 |         "xattn": false,
25 |         "fusedLN": false
26 |     }
27 | }


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/dev_eva_clip/eva_clip/model_configs/EVA-CLIP-8B.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "embed_dim": 1280,
 3 |     "vision_cfg": {
 4 |         "image_size": 224,
 5 |         "layers": 32,
 6 |         "width": 4096,
 7 |         "head_width": 128,
 8 |         "mlp_ratio": 5,
 9 |         "patch_size": 14,
10 |         "eva_model_name": "eva-clip-8b-14-x",
11 |         "drop_path_rate": 0,
12 |         "qkv_bias": false,
13 |         "xattn": true,
14 |         "postnorm": false,
15 |         "fusedLN": false,
16 |         "use_rms_norm": true
17 |     },
18 |     "text_cfg": {
19 |         "context_length": 77,
20 |         "vocab_size": 49408,
21 |         "width": 1280,
22 |         "heads": 20,
23 |         "layers": 32,
24 |         "xattn": false,
25 |         "fusedLN": false
26 |     }
27 | }


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/eva_clip/model_configs/EVA-CLIP-8B-plus.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "embed_dim": 1280,
 3 |     "vision_cfg": {
 4 |         "image_size": 448,
 5 |         "layers": 32,
 6 |         "width": 4096,
 7 |         "head_width": 128,
 8 |         "mlp_ratio": 5,
 9 |         "patch_size": 14,
10 |         "eva_model_name": "eva-clip-8b-14-plus-x",
11 |         "drop_path_rate": 0,
12 |         "qkv_bias": false,
13 |         "xattn": true,
14 |         "postnorm": false,
15 |         "fusedLN": false,
16 |         "use_rms_norm": true
17 |     },
18 |     "text_cfg": {
19 |         "context_length": 77,
20 |         "vocab_size": 49408,
21 |         "width": 1280,
22 |         "heads": 20,
23 |         "layers": 32,
24 |         "xattn": false,
25 |         "fusedLN": false
26 |     }
27 | }


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/dev_eva_clip/eva_clip/model_configs/EVA-CLIP-18B.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "embed_dim": 1536,
 3 |     "vision_cfg": {
 4 |         "image_size": 224,
 5 |         "layers": 48,
 6 |         "width": 5120,
 7 |         "head_width": 128,
 8 |         "mlp_ratio": 5,
 9 |         "patch_size": 14,
10 |         "eva_model_name": "eva-clip-18b-14-x",
11 |         "drop_path_rate": 0,
12 |         "qkv_bias": false,
13 |         "xattn": true,
14 |         "postnorm": true,
15 |         "fusedLN": false,
16 |         "use_rms_norm": true
17 |     },
18 |     "text_cfg": {
19 |         "context_length": 77,
20 |         "vocab_size": 49408,
21 |         "width": 1280,
22 |         "heads": 20,
23 |         "layers": 32,
24 |         "xattn": false,
25 |         "fusedLN": false
26 |     }
27 | }


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/dev_eva_clip/eva_clip/model_configs/EVA-CLIP-8B-plus.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "embed_dim": 1280,
 3 |     "vision_cfg": {
 4 |         "image_size": 448,
 5 |         "layers": 32,
 6 |         "width": 4096,
 7 |         "head_width": 128,
 8 |         "mlp_ratio": 5,
 9 |         "patch_size": 14,
10 |         "eva_model_name": "eva-clip-8b-14-plus-x",
11 |         "drop_path_rate": 0,
12 |         "qkv_bias": false,
13 |         "xattn": true,
14 |         "postnorm": false,
15 |         "fusedLN": false,
16 |         "use_rms_norm": true
17 |     },
18 |     "text_cfg": {
19 |         "context_length": 77,
20 |         "vocab_size": 49408,
21 |         "width": 1280,
22 |         "heads": 20,
23 |         "layers": 32,
24 |         "xattn": false,
25 |         "fusedLN": false
26 |     }
27 | }


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/eva_clip/model_configs/EVA02-CLIP-B-16.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "embed_dim": 512,
 3 |     "vision_cfg": {
 4 |         "image_size": 224,
 5 |         "layers": 12,
 6 |         "width": 768,
 7 |         "head_width": 64,
 8 |         "patch_size": 16,
 9 |         "mlp_ratio": 2.6667,
10 |         "eva_model_name": "eva-clip-b-16-X",
11 |         "drop_path_rate": 0.0,
12 |         "xattn": true,
13 |         "fusedLN": true,
14 |         "rope": true,
15 |         "pt_hw_seq_len": 16,
16 |         "intp_freq": true,
17 |         "naiveswiglu": true,
18 |         "subln": true
19 |     },
20 |     "text_cfg": {
21 |         "context_length": 77,
22 |         "vocab_size": 49408,
23 |         "width": 512,
24 |         "heads": 8,
25 |         "layers": 12,
26 |         "xattn": true,
27 |         "fusedLN": true
28 |     }
29 | }


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/eva_clip/model_configs/EVA02-CLIP-L-14.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "embed_dim": 768,
 3 |     "vision_cfg": {
 4 |         "image_size": 224,
 5 |         "layers": 24,
 6 |         "width": 1024,
 7 |         "drop_path_rate": 0,
 8 |         "head_width": 64,
 9 |         "mlp_ratio": 2.6667,
10 |         "patch_size": 14,
11 |         "eva_model_name": "eva-clip-l-14",
12 |         "xattn": true,
13 |         "fusedLN": true,
14 |         "rope": true,
15 |         "pt_hw_seq_len": 16,
16 |         "intp_freq": true,
17 |         "naiveswiglu": true,
18 |         "subln": true
19 |     },
20 |     "text_cfg": {
21 |         "context_length": 77,
22 |         "vocab_size": 49408,
23 |         "width": 768,
24 |         "heads": 12,
25 |         "layers": 12,
26 |         "xattn": false,
27 |         "fusedLN": true
28 |     }
29 | }


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/eva_clip/model_configs/EVA02-CLIP-L-14-336.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "embed_dim": 768,
 3 |     "vision_cfg": {
 4 |         "image_size": 336,
 5 |         "layers": 24,
 6 |         "width": 1024,
 7 |         "drop_path_rate": 0,
 8 |         "head_width": 64,
 9 |         "mlp_ratio": 2.6667,
10 |         "patch_size": 14,
11 |         "eva_model_name": "eva-clip-l-14-336",
12 |         "xattn": true,
13 |         "fusedLN": true,
14 |         "rope": true,
15 |         "pt_hw_seq_len": 16,
16 |         "intp_freq": true,
17 |         "naiveswiglu": true,
18 |         "subln": true
19 |     },
20 |     "text_cfg": {
21 |         "context_length": 77,
22 |         "vocab_size": 49408,
23 |         "width": 768,
24 |         "heads": 12,
25 |         "layers": 12,
26 |         "xattn": false,
27 |         "fusedLN": true
28 |     }
29 | }


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/groundingdino/models/__init__.py:
--------------------------------------------------------------------------------
 1 | # ------------------------------------------------------------------------
 2 | # Grounding DINO
 3 | # url: https://github.com/IDEA-Research/GroundingDINO
 4 | # Copyright (c) 2023 IDEA. All Rights Reserved.
 5 | # Licensed under the Apache License, Version 2.0 [see LICENSE for details]
 6 | # ------------------------------------------------------------------------
 7 | # Copyright (c) Facebook, Inc. and its affiliates. All Rights Reserved
 8 | from .GroundingDINO import build_groundingdino
 9 | 
10 | 
11 | def build_model(args):
12 |     # we use register to maintain models from catdet6 on.
13 |     from .registry import MODULE_BUILD_FUNCS
14 | 
15 |     assert args.modelname in MODULE_BUILD_FUNCS._module_dict
16 |     build_func = MODULE_BUILD_FUNCS.get(args.modelname)
17 |     model = build_func(args)
18 |     return model
19 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/dev_eva_clip/eva_clip/model_configs/EVA02-CLIP-B-16.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "embed_dim": 512,
 3 |     "vision_cfg": {
 4 |         "image_size": 224,
 5 |         "layers": 12,
 6 |         "width": 768,
 7 |         "head_width": 64,
 8 |         "patch_size": 16,
 9 |         "mlp_ratio": 2.6667,
10 |         "eva_model_name": "eva-clip-b-16-X",
11 |         "drop_path_rate": 0.0,
12 |         "xattn": true,
13 |         "fusedLN": true,
14 |         "rope": true,
15 |         "pt_hw_seq_len": 16,
16 |         "intp_freq": true,
17 |         "naiveswiglu": true,
18 |         "subln": true
19 |     },
20 |     "text_cfg": {
21 |         "context_length": 77,
22 |         "vocab_size": 49408,
23 |         "width": 512,
24 |         "heads": 8,
25 |         "layers": 12,
26 |         "xattn": true,
27 |         "fusedLN": true
28 |     }
29 | }


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/dev_eva_clip/eva_clip/model_configs/EVA02-CLIP-L-14.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "embed_dim": 768,
 3 |     "vision_cfg": {
 4 |         "image_size": 224,
 5 |         "layers": 24,
 6 |         "width": 1024,
 7 |         "drop_path_rate": 0,
 8 |         "head_width": 64,
 9 |         "mlp_ratio": 2.6667,
10 |         "patch_size": 14,
11 |         "eva_model_name": "eva-clip-l-14",
12 |         "xattn": true,
13 |         "fusedLN": true,
14 |         "rope": true,
15 |         "pt_hw_seq_len": 16,
16 |         "intp_freq": true,
17 |         "naiveswiglu": true,
18 |         "subln": true
19 |     },
20 |     "text_cfg": {
21 |         "context_length": 77,
22 |         "vocab_size": 49408,
23 |         "width": 768,
24 |         "heads": 12,
25 |         "layers": 12,
26 |         "xattn": false,
27 |         "fusedLN": true
28 |     }
29 | }


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/dev_eva_clip/eva_clip/model_configs/EVA02-CLIP-L-14-336.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "embed_dim": 768,
 3 |     "vision_cfg": {
 4 |         "image_size": 336,
 5 |         "layers": 24,
 6 |         "width": 1024,
 7 |         "drop_path_rate": 0,
 8 |         "head_width": 64,
 9 |         "mlp_ratio": 2.6667,
10 |         "patch_size": 14,
11 |         "eva_model_name": "eva-clip-l-14-336",
12 |         "xattn": true,
13 |         "fusedLN": true,
14 |         "rope": true,
15 |         "pt_hw_seq_len": 16,
16 |         "intp_freq": true,
17 |         "naiveswiglu": true,
18 |         "subln": true
19 |     },
20 |     "text_cfg": {
21 |         "context_length": 77,
22 |         "vocab_size": 49408,
23 |         "width": 768,
24 |         "heads": 12,
25 |         "layers": 12,
26 |         "xattn": false,
27 |         "fusedLN": true
28 |     }
29 | }


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/dev_eva_clip/eva_clip/__init__.py:
--------------------------------------------------------------------------------
 1 | from .constants import OPENAI_DATASET_MEAN, OPENAI_DATASET_STD
 2 | from .factory import create_model, create_model_and_transforms, create_model_from_pretrained, get_tokenizer
 3 | from .factory import list_models, add_model_config, get_model_config, load_checkpoint
 4 | from .loss import ClipLoss
 5 | from .model import CLIP, CustomCLIP, CLIPTextCfg, CLIPVisionCfg, convert_weights_to_lp, convert_weights_to_fp16, trace_model, get_cast_dtype
 6 | from .openai import load_openai_model, list_openai_models
 7 | from .pretrained import list_pretrained, list_pretrained_models_by_tag, list_pretrained_tags_by_model, get_pretrained_url, download_pretrained_from_url, is_pretrained_cfg, get_pretrained_cfg, download_pretrained
 8 | from .tokenizer import SimpleTokenizer, tokenize
 9 | from .transform import image_transform
10 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/serve/register_worker.py:
--------------------------------------------------------------------------------
 1 | """
 2 | Manually register workers.
 3 | 
 4 | Usage:
 5 | python3 -m fastchat.serve.register_worker --controller http://localhost:21001 --worker-name http://localhost:21002
 6 | """
 7 | 
 8 | import argparse
 9 | 
10 | import requests
11 | 
12 | if __name__ == "__main__":
13 |     parser = argparse.ArgumentParser()
14 |     parser.add_argument("--controller-address", type=str)
15 |     parser.add_argument("--worker-name", type=str)
16 |     parser.add_argument("--check-heart-beat", action="store_true")
17 |     args = parser.parse_args()
18 | 
19 |     url = args.controller_address + "/register_worker"
20 |     data = {
21 |         "worker_name": args.worker_name,
22 |         "check_heart_beat": args.check_heart_beat,
23 |         "worker_status": None,
24 |     }
25 |     r = requests.post(url, json=data)
26 |     assert r.status_code == 200
27 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/groundingdino/models/GroundingDINO/__init__.py:
--------------------------------------------------------------------------------
 1 | # ------------------------------------------------------------------------
 2 | # Grounding DINO
 3 | # url: https://github.com/IDEA-Research/GroundingDINO
 4 | # Copyright (c) 2023 IDEA. All Rights Reserved.
 5 | # Licensed under the Apache License, Version 2.0 [see LICENSE for details]
 6 | # ------------------------------------------------------------------------
 7 | # Conditional DETR
 8 | # Copyright (c) 2021 Microsoft. All Rights Reserved.
 9 | # Licensed under the Apache License, Version 2.0 [see LICENSE for details]
10 | # ------------------------------------------------------------------------
11 | # Copied from DETR (https://github.com/facebookresearch/detr)
12 | # Copyright (c) Facebook, Inc. and its affiliates. All Rights Reserved.
13 | # ------------------------------------------------------------------------
14 | 
15 | from .groundingdino import build_groundingdino
16 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/utils.py:
--------------------------------------------------------------------------------
 1 | from transformers import AutoConfig
 2 | 
 3 | 
 4 | def auto_upgrade(config):
 5 |     cfg = AutoConfig.from_pretrained(config)
 6 |     if "llava" in config and "llava" not in cfg.model_type:
 7 |         assert cfg.model_type == "llama"
 8 |         print("You are using newer LLaVA code base, while the checkpoint of v0 is from older code base.")
 9 |         print("You must upgrade the checkpoint to the new code base (this can be done automatically).")
10 |         confirm = input("Please confirm that you want to upgrade the checkpoint. [Y/N]")
11 |         if confirm.lower() in ["y", "yes"]:
12 |             print("Upgrading checkpoint...")
13 |             assert len(cfg.architectures) == 1
14 |             setattr(cfg.__class__, "model_type", "llava")
15 |             cfg.architectures[0] = "LlavaLlamaForCausalLM"
16 |             cfg.save_pretrained(config)
17 |             print("Checkpoint upgraded.")
18 |         else:
19 |             print("Checkpoint upgrade aborted.")
20 |             exit(1)
21 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/consolidate.py:
--------------------------------------------------------------------------------
 1 | """
 2 | Usage:
 3 | python3 -m llava.model.consolidate --src ~/model_weights/llava-7b --dst ~/model_weights/llava-7b_consolidate
 4 | """
 5 | 
 6 | import argparse
 7 | 
 8 | import torch
 9 | from transformers import AutoTokenizer, AutoModelForCausalLM
10 | from llava.model import *
11 | from llava.model.utils import auto_upgrade
12 | 
13 | 
14 | def consolidate_ckpt(src_path, dst_path):
15 |     print("Loading model")
16 |     auto_upgrade(src_path)
17 |     src_model = AutoModelForCausalLM.from_pretrained(src_path, torch_dtype=torch.float16, low_cpu_mem_usage=True)
18 |     src_tokenizer = AutoTokenizer.from_pretrained(src_path, use_fast=False)
19 |     src_model.save_pretrained(dst_path)
20 |     src_tokenizer.save_pretrained(dst_path)
21 | 
22 | 
23 | if __name__ == "__main__":
24 |     parser = argparse.ArgumentParser()
25 |     parser.add_argument("--src", type=str, required=True)
26 |     parser.add_argument("--dst", type=str, required=True)
27 | 
28 |     args = parser.parse_args()
29 | 
30 |     consolidate_ckpt(args.src, args.dst)
31 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_projector/pooler_projector.py:
--------------------------------------------------------------------------------
 1 | import torch
 2 | import torch.nn as nn
 3 | 
 4 | import math
 5 | 
 6 | from transformers.models.clip.modeling_clip import CLIPVisionModel
 7 | 
 8 | 
 9 | class PoolerProjector(nn.Module):
10 |     def __init__(self, config, vision_cfg):
11 |         super().__init__()
12 |         self._config = config
13 |         self.hw = vision_cfg.image_size // vision_cfg.patch_size
14 | 
15 |         self.conv_pool = nn.Conv2d(config.mm_hidden_size, config.hidden_size, kernel_size=2, stride=2)
16 | 
17 |         self.proj = nn.Sequential(
18 |             nn.GELU(),
19 |             nn.Linear(config.hidden_size, config.hidden_size),
20 |         )
21 | 
22 |     def forward(self, x, *args, **kwargs):
23 |         height = width = self.hw
24 |         assert height * width == x.shape[1]
25 |         x = x.view(x.shape[0], height, width, -1).permute(0, 3, 1, 2)
26 |         x = self.conv_pool(x)
27 |         x = x.flatten(2).transpose(1, 2)
28 |         x = self.proj(x)
29 |         return x
30 | 
31 |     @property
32 |     def config(self):
33 |         return {"mm_projector_type": "pooler"}
34 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/janus/utils/__init__.py:
--------------------------------------------------------------------------------
 1 | # Copyright (c) 2023-2024 DeepSeek.
 2 | #
 3 | # Permission is hereby granted, free of charge, to any person obtaining a copy of
 4 | # this software and associated documentation files (the "Software"), to deal in
 5 | # the Software without restriction, including without limitation the rights to
 6 | # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
 7 | # the Software, and to permit persons to whom the Software is furnished to do so,
 8 | # subject to the following conditions:
 9 | #
10 | # The above copyright notice and this permission notice shall be included in all
11 | # copies or substantial portions of the Software.
12 | #
13 | # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
14 | # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
15 | # FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
16 | # COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
17 | # IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
18 | # CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
19 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_resampler/builder.py:
--------------------------------------------------------------------------------
 1 | import torch
 2 | 
 3 | from .masked_drop import MaskedDrop
 4 | from .spatial_pool import SpatialPool
 5 | from .perceiver import PerceiverResampler
 6 | from .qformer import Qformer
 7 | 
 8 | 
 9 | class IdentityMap(torch.nn.Module):
10 |     def __init__(self):
11 |         super().__init__()
12 | 
13 |     def forward(self, x, *args, **kwargs):
14 |         return x
15 | 
16 |     @property
17 |     def config(self):
18 |         return {"mm_resampler_type": None}
19 | 
20 | 
21 | def build_vision_resampler(model_args, delay_load=False, **kwargs):
22 |     resampler_type = getattr(model_args, "mm_resampler_type", None)
23 |     if resampler_type == "masked_drop":
24 |         return MaskedDrop(model_args)
25 |     elif resampler_type == "spatial_pool":
26 |         return SpatialPool(model_args, **kwargs)
27 |     elif resampler_type == "perceiver":
28 |         return PerceiverResampler(model_args, **kwargs)
29 |     elif resampler_type == "qformer":
30 |         return Qformer(model_args, **kwargs)
31 |     elif resampler_type is None:
32 |         return IdentityMap()
33 | 
34 |     raise ValueError(f"Unknown resampler type: {resampler_type}")
35 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/Dockerfile:
--------------------------------------------------------------------------------
 1 | FROM pytorch/pytorch:2.1.2-cuda12.1-cudnn8-runtime
 2 | ARG DEBIAN_FRONTEND=noninteractive
 3 | 
 4 | ENV CUDA_HOME=/usr/local/cuda \
 5 |      TORCH_CUDA_ARCH_LIST="6.0 6.1 7.0 7.5 8.0 8.6+PTX" \
 6 |      SETUPTOOLS_USE_DISTUTILS=stdlib
 7 | 
 8 | RUN conda update conda -y
 9 | 
10 | # Install libraries in the brand new image. 
11 | RUN apt-get -y update && apt-get install -y --no-install-recommends \
12 |          wget \
13 |          build-essential \
14 |          git \
15 |          python3-opencv \
16 |          ca-certificates && \
17 |     rm -rf /var/lib/apt/lists/*
18 | 
19 | # Set the working directory for all the subsequent Dockerfile instructions.
20 | WORKDIR /opt/program
21 | 
22 | RUN git clone https://github.com/IDEA-Research/GroundingDINO.git
23 | 
24 | RUN mkdir weights ; cd weights ; wget -q https://github.com/IDEA-Research/GroundingDINO/releases/download/v0.1.0-alpha/groundingdino_swint_ogc.pth ; cd ..
25 | 
26 | RUN conda install -c "nvidia/label/cuda-12.1.1" cuda -y
27 | ENV CUDA_HOME=$CONDA_PREFIX
28 | 
29 | ENV PATH=/usr/local/cuda/bin:$PATH
30 | 
31 | RUN cd GroundingDINO/ && python -m pip install .
32 | 
33 | COPY docker_test.py docker_test.py
34 | 
35 | CMD [ "python", "docker_test.py" ]


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/groundingdino/config/GroundingDINO_SwinB_cfg.py:
--------------------------------------------------------------------------------
 1 | batch_size = 1
 2 | modelname = "groundingdino"
 3 | backbone = "swin_B_384_22k"
 4 | position_embedding = "sine"
 5 | pe_temperatureH = 20
 6 | pe_temperatureW = 20
 7 | return_interm_indices = [1, 2, 3]
 8 | backbone_freeze_keywords = None
 9 | enc_layers = 6
10 | dec_layers = 6
11 | pre_norm = False
12 | dim_feedforward = 2048
13 | hidden_dim = 256
14 | dropout = 0.0
15 | nheads = 8
16 | num_queries = 900
17 | query_dim = 4
18 | num_patterns = 0
19 | num_feature_levels = 4
20 | enc_n_points = 4
21 | dec_n_points = 4
22 | two_stage_type = "standard"
23 | two_stage_bbox_embed_share = False
24 | two_stage_class_embed_share = False
25 | transformer_activation = "relu"
26 | dec_pred_bbox_embed_share = True
27 | dn_box_noise_scale = 1.0
28 | dn_label_noise_ratio = 0.5
29 | dn_label_coef = 1.0
30 | dn_bbox_coef = 1.0
31 | embed_init_tgt = True
32 | dn_labelbook_size = 2000
33 | max_text_len = 256
34 | text_encoder_type = "bert-base-uncased"
35 | use_text_enhancer = True
36 | use_fusion_layer = True
37 | use_checkpoint = True
38 | use_transformer_ckpt = True
39 | use_text_cross_attention = True
40 | text_dropout = 0.0
41 | fusion_dropout = 0.0
42 | fusion_droppath = 0.1
43 | sub_sentence_present = True
44 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/groundingdino/config/GroundingDINO_SwinT_OGC.py:
--------------------------------------------------------------------------------
 1 | batch_size = 1
 2 | modelname = "groundingdino"
 3 | backbone = "swin_T_224_1k"
 4 | position_embedding = "sine"
 5 | pe_temperatureH = 20
 6 | pe_temperatureW = 20
 7 | return_interm_indices = [1, 2, 3]
 8 | backbone_freeze_keywords = None
 9 | enc_layers = 6
10 | dec_layers = 6
11 | pre_norm = False
12 | dim_feedforward = 2048
13 | hidden_dim = 256
14 | dropout = 0.0
15 | nheads = 8
16 | num_queries = 900
17 | query_dim = 4
18 | num_patterns = 0
19 | num_feature_levels = 4
20 | enc_n_points = 4
21 | dec_n_points = 4
22 | two_stage_type = "standard"
23 | two_stage_bbox_embed_share = False
24 | two_stage_class_embed_share = False
25 | transformer_activation = "relu"
26 | dec_pred_bbox_embed_share = True
27 | dn_box_noise_scale = 1.0
28 | dn_label_noise_ratio = 0.5
29 | dn_label_coef = 1.0
30 | dn_bbox_coef = 1.0
31 | embed_init_tgt = True
32 | dn_labelbook_size = 2000
33 | max_text_len = 256
34 | text_encoder_type = "bert-base-uncased"
35 | use_text_enhancer = True
36 | use_fusion_layer = True
37 | use_checkpoint = True
38 | use_transformer_ckpt = True
39 | use_text_cross_attention = True
40 | text_dropout = 0.0
41 | fusion_dropout = 0.0
42 | fusion_droppath = 0.1
43 | sub_sentence_present = True
44 | 


--------------------------------------------------------------------------------
/src/t2i-r1/configs/zero3.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "fp16": {
 3 |         "enabled": "auto",
 4 |         "loss_scale": 0,
 5 |         "loss_scale_window": 1000,
 6 |         "initial_scale_power": 16,
 7 |         "hysteresis": 2,
 8 |         "min_loss_scale": 1
 9 |     },
10 |     "bf16": {
11 |         "enabled": "auto"
12 |     },
13 | 
14 |     "zero_optimization": {
15 |         "stage": 3,
16 |         "offload_optimizer": {
17 |             "device": "none",
18 |             "pin_memory": true
19 |         },
20 |         "offload_param": {
21 |             "device": "none",
22 |             "pin_memory": true
23 |         },
24 |         "overlap_comm": true,
25 |         "contiguous_gradients": true,
26 |         "sub_group_size": 1e9,
27 |         "reduce_bucket_size": "auto",
28 |         "stage3_prefetch_bucket_size": "auto",
29 |         "stage3_param_persistence_threshold": "auto",
30 |         "stage3_max_live_parameters": 1e9,
31 |         "stage3_max_reuse_distance": 1e9,
32 |         "stage3_gather_16bit_weights_on_model_save": true
33 |     },
34 | 
35 |     "gradient_accumulation_steps": "auto",
36 |     "gradient_clipping": "auto",
37 |     "steps_per_print": 100,
38 |     "train_batch_size": "auto",
39 |     "train_micro_batch_size_per_gpu": "auto",
40 |     "wall_clock_breakdown": false
41 | }


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/groundingdino/models/GroundingDINO/csrc/MsDeformAttn/ms_deform_attn_cuda.h:
--------------------------------------------------------------------------------
 1 | /*!
 2 | **************************************************************************************************
 3 | * Deformable DETR
 4 | * Copyright (c) 2020 SenseTime. All Rights Reserved.
 5 | * Licensed under the Apache License, Version 2.0 [see LICENSE for details]
 6 | **************************************************************************************************
 7 | * Modified from https://github.com/chengdazhi/Deformable-Convolution-V2-PyTorch/tree/pytorch_1.0.0
 8 | **************************************************************************************************
 9 | */
10 | 
11 | #pragma once
12 | #include <torch/extension.h>
13 | 
14 | namespace groundingdino {
15 | 
16 | at::Tensor ms_deform_attn_cuda_forward(
17 |     const at::Tensor &value, 
18 |     const at::Tensor &spatial_shapes,
19 |     const at::Tensor &level_start_index,
20 |     const at::Tensor &sampling_loc,
21 |     const at::Tensor &attn_weight,
22 |     const int im2col_step);
23 | 
24 | std::vector<at::Tensor> ms_deform_attn_cuda_backward(
25 |     const at::Tensor &value, 
26 |     const at::Tensor &spatial_shapes,
27 |     const at::Tensor &level_start_index,
28 |     const at::Tensor &sampling_loc,
29 |     const at::Tensor &attn_weight,
30 |     const at::Tensor &grad_output,
31 |     const int im2col_step);
32 | 
33 | } // namespace groundingdino


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/groundingdino/models/GroundingDINO/csrc/MsDeformAttn/ms_deform_attn_cpu.h:
--------------------------------------------------------------------------------
 1 | /*!
 2 | **************************************************************************************************
 3 | * Deformable DETR
 4 | * Copyright (c) 2020 SenseTime. All Rights Reserved.
 5 | * Licensed under the Apache License, Version 2.0 [see LICENSE for details]
 6 | **************************************************************************************************
 7 | * Modified from https://github.com/chengdazhi/Deformable-Convolution-V2-PyTorch/tree/pytorch_1.0.0
 8 | **************************************************************************************************
 9 | */
10 | 
11 | #pragma once
12 | #include <torch/extension.h>
13 | 
14 | namespace groundingdino {
15 | 
16 | at::Tensor
17 | ms_deform_attn_cpu_forward(
18 |     const at::Tensor &value, 
19 |     const at::Tensor &spatial_shapes,
20 |     const at::Tensor &level_start_index,
21 |     const at::Tensor &sampling_loc,
22 |     const at::Tensor &attn_weight,
23 |     const int im2col_step);
24 | 
25 | std::vector<at::Tensor>
26 | ms_deform_attn_cpu_backward(
27 |     const at::Tensor &value, 
28 |     const at::Tensor &spatial_shapes,
29 |     const at::Tensor &level_start_index,
30 |     const at::Tensor &sampling_loc,
31 |     const at::Tensor &attn_weight,
32 |     const at::Tensor &grad_output,
33 |     const int im2col_step);
34 | 
35 | } // namespace groundingdino
36 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/janus/models/__init__.py:
--------------------------------------------------------------------------------
 1 | # Copyright (c) 2023-2024 DeepSeek.
 2 | #
 3 | # Permission is hereby granted, free of charge, to any person obtaining a copy of
 4 | # this software and associated documentation files (the "Software"), to deal in
 5 | # the Software without restriction, including without limitation the rights to
 6 | # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
 7 | # the Software, and to permit persons to whom the Software is furnished to do so,
 8 | # subject to the following conditions:
 9 | #
10 | # The above copyright notice and this permission notice shall be included in all
11 | # copies or substantial portions of the Software.
12 | #
13 | # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
14 | # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
15 | # FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
16 | # COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
17 | # IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
18 | # CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
19 | 
20 | from .image_processing_vlm import VLMImageProcessor
21 | from .modeling_vlm import MultiModalityCausalLM
22 | from .processing_vlm import VLChatProcessor
23 | 
24 | __all__ = [
25 |     "VLMImageProcessor",
26 |     "VLChatProcessor",
27 |     "MultiModalityCausalLM",
28 | ]
29 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/janus/janusflow/models/__init__.py:
--------------------------------------------------------------------------------
 1 | # Copyright (c) 2023-2024 DeepSeek.
 2 | #
 3 | # Permission is hereby granted, free of charge, to any person obtaining a copy of
 4 | # this software and associated documentation files (the "Software"), to deal in
 5 | # the Software without restriction, including without limitation the rights to
 6 | # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
 7 | # the Software, and to permit persons to whom the Software is furnished to do so,
 8 | # subject to the following conditions:
 9 | #
10 | # The above copyright notice and this permission notice shall be included in all
11 | # copies or substantial portions of the Software.
12 | #
13 | # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
14 | # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
15 | # FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
16 | # COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
17 | # IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
18 | # CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
19 | 
20 | from .image_processing_vlm import VLMImageProcessor
21 | from .modeling_vlm import MultiModalityCausalLM
22 | from .processing_vlm import VLChatProcessor
23 | 
24 | __all__ = [
25 |     "VLMImageProcessor",
26 |     "VLChatProcessor",
27 |     "MultiModalityCausalLM",
28 | ]
29 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/groundingdino/util/get_tokenlizer.py:
--------------------------------------------------------------------------------
 1 | from transformers import AutoTokenizer, BertModel, BertTokenizer, RobertaModel, RobertaTokenizerFast
 2 | import os
 3 | 
 4 | def get_tokenlizer(text_encoder_type):
 5 |     if not isinstance(text_encoder_type, str):
 6 |         # print("text_encoder_type is not a str")
 7 |         if hasattr(text_encoder_type, "text_encoder_type"):
 8 |             text_encoder_type = text_encoder_type.text_encoder_type
 9 |         elif text_encoder_type.get("text_encoder_type", False):
10 |             text_encoder_type = text_encoder_type.get("text_encoder_type")
11 |         elif os.path.isdir(text_encoder_type) and os.path.exists(text_encoder_type):
12 |             pass
13 |         else:
14 |             raise ValueError(
15 |                 "Unknown type of text_encoder_type: {}".format(type(text_encoder_type))
16 |             )
17 |     print("final text_encoder_type: {}".format(text_encoder_type))
18 | 
19 |     tokenizer = AutoTokenizer.from_pretrained(text_encoder_type)
20 |     return tokenizer
21 | 
22 | 
23 | def get_pretrained_language_model(text_encoder_type):
24 |     if text_encoder_type == "bert-base-uncased" or (os.path.isdir(text_encoder_type) and os.path.exists(text_encoder_type)):
25 |         return BertModel.from_pretrained(text_encoder_type)
26 |     if text_encoder_type == "roberta-base":
27 |         return RobertaModel.from_pretrained(text_encoder_type)
28 | 
29 |     raise ValueError("Unknown text_encoder_type {}".format(text_encoder_type))
30 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/groundingdino/models/GroundingDINO/csrc/MsDeformAttn/ms_deform_attn_cpu.cpp:
--------------------------------------------------------------------------------
 1 | /*!
 2 | **************************************************************************************************
 3 | * Deformable DETR
 4 | * Copyright (c) 2020 SenseTime. All Rights Reserved.
 5 | * Licensed under the Apache License, Version 2.0 [see LICENSE for details]
 6 | **************************************************************************************************
 7 | * Modified from https://github.com/chengdazhi/Deformable-Convolution-V2-PyTorch/tree/pytorch_1.0.0
 8 | **************************************************************************************************
 9 | */
10 | 
11 | #include <vector>
12 | 
13 | #include <ATen/ATen.h>
14 | #include <ATen/cuda/CUDAContext.h>
15 | 
16 | namespace groundingdino {
17 | 
18 | at::Tensor
19 | ms_deform_attn_cpu_forward(
20 |     const at::Tensor &value, 
21 |     const at::Tensor &spatial_shapes,
22 |     const at::Tensor &level_start_index,
23 |     const at::Tensor &sampling_loc,
24 |     const at::Tensor &attn_weight,
25 |     const int im2col_step)
26 | {
27 |     AT_ERROR("Not implement on cpu");
28 | }
29 | 
30 | std::vector<at::Tensor>
31 | ms_deform_attn_cpu_backward(
32 |     const at::Tensor &value, 
33 |     const at::Tensor &spatial_shapes,
34 |     const at::Tensor &level_start_index,
35 |     const at::Tensor &sampling_loc,
36 |     const at::Tensor &attn_weight,
37 |     const at::Tensor &grad_output,
38 |     const int im2col_step)
39 | {
40 |     AT_ERROR("Not implement on cpu");
41 | }
42 | 
43 | } // namespace groundingdino
44 | 


--------------------------------------------------------------------------------
/src/t2i-r1/configs/zero3_offload.json:
--------------------------------------------------------------------------------
 1 | {
 2 |     "fp16": {
 3 |         "enabled": "auto",
 4 |         "loss_scale": 0,
 5 |         "loss_scale_window": 1000,
 6 |         "initial_scale_power": 16,
 7 |         "hysteresis": 2,
 8 |         "min_loss_scale": 1
 9 |     },
10 |     "bf16": {
11 |         "enabled": "auto"
12 |     },
13 |     "optimizer": {
14 |         "type": "AdamW",
15 |         "params": {
16 |             "lr": "auto",
17 |             "betas": "auto",
18 |             "eps": "auto",
19 |             "weight_decay": "auto"
20 |         }
21 |     },
22 |     "zero_optimization": {
23 |         "stage": 3,
24 |         "offload_optimizer": {
25 |             "device": "cpu",
26 |             "pin_memory": true
27 |         },
28 |         "offload_param": {
29 |             "device": "cpu",
30 |             "pin_memory": true
31 |         },
32 |         "overlap_comm": true,
33 |         "contiguous_gradients": true,
34 |         "sub_group_size": 5e8,
35 |         "reduce_bucket_size": 5e7,
36 |         "stage3_prefetch_bucket_size": 5e7,
37 |         "stage3_param_persistence_threshold": 1e5,
38 |         "stage3_max_live_parameters": 5e8,
39 |         "stage3_max_reuse_distance": 5e8,
40 |         "stage3_gather_16bit_weights_on_model_save": true
41 |     },
42 |     "zero_allow_untested_optimizer": true,
43 |     "gradient_accumulation_steps": "auto",
44 |     "gradient_clipping": "auto",
45 |     "train_batch_size": "auto",
46 |     "train_micro_batch_size_per_gpu": "auto",
47 |     "steps_per_print": 1e5,
48 |     "wall_clock_breakdown": false,
49 |     "memory_breakdown": false
50 | }


--------------------------------------------------------------------------------
/src/t2i-r1/src/janus/__init__.py:
--------------------------------------------------------------------------------
 1 | # Copyright (c) 2023-2024 DeepSeek.
 2 | #
 3 | # Permission is hereby granted, free of charge, to any person obtaining a copy of
 4 | # this software and associated documentation files (the "Software"), to deal in
 5 | # the Software without restriction, including without limitation the rights to
 6 | # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
 7 | # the Software, and to permit persons to whom the Software is furnished to do so,
 8 | # subject to the following conditions:
 9 | #
10 | # The above copyright notice and this permission notice shall be included in all
11 | # copies or substantial portions of the Software.
12 | #
13 | # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
14 | # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
15 | # FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
16 | # COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
17 | # IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
18 | # CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
19 | 
20 | 
21 | # check if python version is above 3.10
22 | import sys
23 | 
24 | if sys.version_info >= (3, 10):
25 |     print("Python version is above 3.10, patching the collections module.")
26 |     # Monkey patch collections
27 |     import collections
28 |     import collections.abc
29 | 
30 |     for type_name in collections.abc.__all__:
31 |         setattr(collections, type_name, getattr(collections.abc, type_name))
32 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/janus/janusflow/__init__.py:
--------------------------------------------------------------------------------
 1 | # Copyright (c) 2023-2024 DeepSeek.
 2 | #
 3 | # Permission is hereby granted, free of charge, to any person obtaining a copy of
 4 | # this software and associated documentation files (the "Software"), to deal in
 5 | # the Software without restriction, including without limitation the rights to
 6 | # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
 7 | # the Software, and to permit persons to whom the Software is furnished to do so,
 8 | # subject to the following conditions:
 9 | #
10 | # The above copyright notice and this permission notice shall be included in all
11 | # copies or substantial portions of the Software.
12 | #
13 | # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
14 | # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
15 | # FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
16 | # COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
17 | # IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
18 | # CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
19 | 
20 | 
21 | # check if python version is above 3.10
22 | import sys
23 | 
24 | if sys.version_info >= (3, 10):
25 |     print("Python version is above 3.10, patching the collections module.")
26 |     # Monkey patch collections
27 |     import collections
28 |     import collections.abc
29 | 
30 |     for type_name in collections.abc.__all__:
31 |         setattr(collections, type_name, getattr(collections.abc, type_name))
32 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/pyproject.toml:
--------------------------------------------------------------------------------
 1 | [tool.black]
 2 | line-length = 240
 3 | 
 4 | [build-system]
 5 | requires = ["setuptools>=61.0"]
 6 | build-backend = "setuptools.build_meta"
 7 | 
 8 | [project]
 9 | name = "llava"
10 | version = "1.7.0.dev0"
11 | description = "LLaVA OneVision: The Next Generation of LLaVA with Better Image and Video Understanding Capabilities"
12 | readme = "README.md"
13 | requires-python = ">=3.8"
14 | classifiers = [
15 |     "Programming Language :: Python :: 3",
16 |     "License :: OSI Approved :: Apache Software License",
17 | ]
18 | 
19 | [project.optional-dependencies]
20 | standalone = [
21 |     "shortuuid",
22 |     "httpx==0.24.0",
23 |     "einops",
24 |     "ftfy",
25 | ]
26 | 
27 | 
28 | train = [
29 |     "llava[standalone]"
30 | ]
31 | 
32 | [project.urls]
33 | "Homepage" = "https://llava-vl.github.io"
34 | "Bug Tracker" = "https://github.com/haotian-liu/LLaVA/issues"
35 | 
36 | [tool.setuptools.packages.find]
37 | include = ["llava*", "trl*"]
38 | exclude = [
39 |     "assets*",
40 |     "benchmark*",
41 |     "docs",
42 |     "dist*",
43 |     "playground*",
44 |     "scripts*",
45 |     "tests*",
46 |     "checkpoints*",
47 |     "project_checkpoints*",
48 |     "debug_checkpoints*",
49 |     "mlx_configs*",
50 |     "wandb*",
51 |     "notebooks*",
52 | ]
53 | 
54 | [tool.wheel]
55 | exclude = [
56 |     "assets*",
57 |     "benchmark*",
58 |     "docs",
59 |     "dist*",
60 |     "playground*",
61 |     "scripts*",
62 |     "tests*",
63 |     "checkpoints*",
64 |     "project_checkpoints*",
65 |     "debug_checkpoints*",
66 |     "mlx_configs*",
67 |     "wandb*",
68 |     "notebooks*",
69 | ]
70 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/groundingdino/models/GroundingDINO/csrc/vision.cpp:
--------------------------------------------------------------------------------
 1 | // Copyright (c) Facebook, Inc. and its affiliates. All Rights Reserved
 2 | 
 3 | #include "MsDeformAttn/ms_deform_attn.h"
 4 | 
 5 | namespace groundingdino {
 6 | 
 7 | #ifdef WITH_CUDA
 8 | extern int get_cudart_version();
 9 | #endif
10 | 
11 | std::string get_cuda_version() {
12 | #ifdef WITH_CUDA
13 |   std::ostringstream oss;
14 | 
15 |   // copied from
16 |   // https://github.com/pytorch/pytorch/blob/master/aten/src/ATen/cuda/detail/CUDAHooks.cpp#L231
17 |   auto printCudaStyleVersion = [&](int v) {
18 |     oss << (v / 1000) << "." << (v / 10 % 100);
19 |     if (v % 10 != 0) {
20 |       oss << "." << (v % 10);
21 |     }
22 |   };
23 |   printCudaStyleVersion(get_cudart_version());
24 |   return oss.str();
25 | #else
26 |   return std::string("not available");
27 | #endif
28 | }
29 | 
30 | // similar to
31 | // https://github.com/pytorch/pytorch/blob/master/aten/src/ATen/Version.cpp
32 | std::string get_compiler_version() {
33 |   std::ostringstream ss;
34 | #if defined(__GNUC__)
35 | #ifndef __clang__
36 |   { ss << "GCC " << __GNUC__ << "." << __GNUC_MINOR__; }
37 | #endif
38 | #endif
39 | 
40 | #if defined(__clang_major__)
41 |   {
42 |     ss << "clang " << __clang_major__ << "." << __clang_minor__ << "."
43 |        << __clang_patchlevel__;
44 |   }
45 | #endif
46 | 
47 | #if defined(_MSC_VER)
48 |   { ss << "MSVC " << _MSC_FULL_VER; }
49 | #endif
50 |   return ss.str();
51 | }
52 | 
53 | PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
54 |   m.def("ms_deform_attn_forward", &ms_deform_attn_forward, "ms_deform_attn_forward");
55 |   m.def("ms_deform_attn_backward", &ms_deform_attn_backward, "ms_deform_attn_backward");
56 | }
57 | 
58 | } // namespace groundingdino


--------------------------------------------------------------------------------
/src/scripts/run_grpo.sh:
--------------------------------------------------------------------------------
 1 | #!/bin/bash
 2 | 
 3 | cd t2i-r1/src
 4 | RUN_NAME="t2i-r1"
 5 | 
 6 | export DEBUG_MODE="true"
 7 | export LOG_PATH="./outputs/debug.txt"
 8 | # export NCCL_DEBUG=INFO
 9 | 
10 | QWEN_PATH="deepseek-ai/Janus-Pro-7B"
11 | HF_DATASET="../../../data/geneval_and_t2i_data_final.json" 
12 | OUTPUT_DIR="janus/outputs/${RUN_NAME}" 
13 | 
14 | PYTHONPATH="$(dirname $0)/..":$PYTHONPATH \
15 | CUDA_VISIBLE_DEVICES="0,1,2,3,4,5,6,7" \
16 | torchrun --nproc_per_node="8" \
17 | --nnodes="1" \
18 | --node_rank="0" \
19 | --master_addr="127.0.0.1" \
20 | --master_port="12345" \
21 | open_r1/grpo.py --use_vllm False \
22 | --deepspeed "../configs/zero3.json" \
23 | --output_dir $OUTPUT_DIR \
24 | --model_name_or_path $QWEN_PATH \
25 | --dataset_name $HF_DATASET \
26 | --max_prompt_length 512 \
27 | --max_completion_length 1024 \
28 | --temperature 1.0 \
29 | --num_generations 8 \
30 | --per_device_train_batch_size 1 \
31 | --gradient_accumulation_steps 2 \
32 | --logging_steps 1 \
33 | --bf16  \
34 | --torch_dtype bfloat16 \
35 | --report_to wandb \
36 | --gradient_checkpointing false \
37 | --attn_implementation flash_attention_2 \
38 | --max_steps 1600 \
39 | --run_name $RUN_NAME \
40 | --save_steps 400 \
41 | --new_generations_image 1 \
42 | --image_token_num_per_image 576 \
43 | --cfg_weight 5 \
44 | --reasoning_prompt_path ../../../data/prompt/reasoning_prompt.txt \
45 | --reward_funcs hps git gdino orm \
46 | --beta 0.01 \
47 | --tf32 true \
48 | --learning_rate 1e-6 \
49 | --hps_ckpt_path ../../../reward_weight/HPS_v2.1_compressed.pt \
50 | --git_ckpt_path ../../../reward_weight/git-large-vqav2 \
51 | --gdino_ckpt_path ../../../reward_weight/groundingdino_swint_ogc.pth \
52 | --gdino_config_path utils/GroundingDINO/groundingdino/config/GroundingDINO_SwinT_OGC.py \
53 | --orm_ckpt_path ../../../reward_weight/ORM-T2I-R1 \
54 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/groundingdino/util/time_counter.py:
--------------------------------------------------------------------------------
 1 | import json
 2 | import time
 3 | 
 4 | 
 5 | class TimeCounter:
 6 |     def __init__(self) -> None:
 7 |         pass
 8 | 
 9 |     def clear(self):
10 |         self.timedict = {}
11 |         self.basetime = time.perf_counter()
12 | 
13 |     def timeit(self, name):
14 |         nowtime = time.perf_counter() - self.basetime
15 |         self.timedict[name] = nowtime
16 |         self.basetime = time.perf_counter()
17 | 
18 | 
19 | class TimeHolder:
20 |     def __init__(self) -> None:
21 |         self.timedict = {}
22 | 
23 |     def update(self, _timedict: dict):
24 |         for k, v in _timedict.items():
25 |             if k not in self.timedict:
26 |                 self.timedict[k] = AverageMeter(name=k, val_only=True)
27 |             self.timedict[k].update(val=v)
28 | 
29 |     def final_res(self):
30 |         return {k: v.avg for k, v in self.timedict.items()}
31 | 
32 |     def __str__(self):
33 |         return json.dumps(self.final_res(), indent=2)
34 | 
35 | 
36 | class AverageMeter(object):
37 |     """Computes and stores the average and current value"""
38 | 
39 |     def __init__(self, name, fmt=":f", val_only=False):
40 |         self.name = name
41 |         self.fmt = fmt
42 |         self.val_only = val_only
43 |         self.reset()
44 | 
45 |     def reset(self):
46 |         self.val = 0
47 |         self.avg = 0
48 |         self.sum = 0
49 |         self.count = 0
50 | 
51 |     def update(self, val, n=1):
52 |         self.val = val
53 |         self.sum += val * n
54 |         self.count += n
55 |         self.avg = self.sum / self.count
56 | 
57 |     def __str__(self):
58 |         if self.val_only:
59 |             fmtstr = "{name} {val" + self.fmt + "}"
60 |         else:
61 |             fmtstr = "{name} {val" + self.fmt + "} ({avg" + self.fmt + "})"
62 |         return fmtstr.format(**self.__dict__)
63 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_resampler/spatial_pool.py:
--------------------------------------------------------------------------------
 1 | import torch
 2 | import torch.nn as nn
 3 | import math
 4 | 
 5 | 
 6 | class SpatialPool(nn.Module):
 7 |     def __init__(self, model_args, vision_tower):
 8 |         super().__init__()
 9 | 
10 |         self.mode = model_args.mm_spatial_pool_mode
11 |         self.stride = model_args.mm_spatial_pool_stride
12 |         self.out_channels = getattr(model_args, "mm_spatial_pool_out_channels", vision_tower.hidden_size)
13 | 
14 |         if self.mode == "average":
15 |             self.pool = nn.AvgPool2d(kernel_size=self.stride, stride=self.stride)
16 |         elif self.mode == "max":
17 |             self.pool = nn.MaxPool2d(kernel_size=self.stride, stride=self.stride)
18 |         elif self.mode == "conv":
19 |             self.pool = nn.Conv2d(in_channels=vision_tower.hidden_size, out_channels=self.out_channels, kernel_size=self.stride, stride=self.stride)
20 |         else:
21 |             raise ValueError(f"Unknown pooling mode: {self.pool}.")
22 | 
23 |     def forward(self, image_features, images, *args, **kwargs):
24 |         ori_W = int(math.sqrt(image_features.shape[1] * images.shape[3] // images.shape[2]))
25 |         ori_H = int(ori_W * images.shape[2] // images.shape[3])
26 | 
27 |         B, _, F = image_features.shape
28 | 
29 |         image_features_spatial = image_features.view(B, ori_H, ori_H, F).permute(0, 3, 1, 2)
30 |         image_features_spatial_pool = self.pool(image_features_spatial)
31 | 
32 |         return image_features_spatial_pool.flatten(2).transpose(1, 2).contiguous()
33 | 
34 |     @property
35 |     def config(self):
36 |         return {
37 |             "mm_resampler_type": "spatial_pool",
38 |             "mm_spatial_pool_stride": self.stride,
39 |             "mm_spatial_pool_mode": self.mode,
40 |             "mm_spatial_pool_out_channels": self.out_channels,
41 |         }
42 | 
43 |     @property
44 |     def hidden_size(self):
45 |         return self.out_channels
46 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/eva_clip/factory.py:
--------------------------------------------------------------------------------
 1 | import json
 2 | import logging
 3 | import os
 4 | import pathlib
 5 | import re
 6 | from copy import deepcopy
 7 | from pathlib import Path
 8 | from typing import Optional, Tuple, Union, Dict, Any
 9 | import torch
10 | 
11 | _MODEL_CONFIG_PATHS = [Path(__file__).parent / f"model_configs/"]
12 | _MODEL_CONFIGS = {}  # directory (model_name: config) of model architecture configs
13 | 
14 | 
15 | def _natural_key(string_):
16 |     return [int(s) if s.isdigit() else s for s in re.split(r"(\d+)", string_.lower())]
17 | 
18 | 
19 | def _rescan_model_configs():
20 |     global _MODEL_CONFIGS
21 | 
22 |     config_ext = (".json",)
23 |     config_files = []
24 |     for config_path in _MODEL_CONFIG_PATHS:
25 |         if config_path.is_file() and config_path.suffix in config_ext:
26 |             config_files.append(config_path)
27 |         elif config_path.is_dir():
28 |             for ext in config_ext:
29 |                 config_files.extend(config_path.glob(f"*{ext}"))
30 | 
31 |     for cf in config_files:
32 |         with open(cf, "r", encoding="utf8") as f:
33 |             model_cfg = json.load(f)
34 |             if all(a in model_cfg for a in ("embed_dim", "vision_cfg", "text_cfg")):
35 |                 _MODEL_CONFIGS[cf.stem] = model_cfg
36 | 
37 |     _MODEL_CONFIGS = dict(sorted(_MODEL_CONFIGS.items(), key=lambda x: _natural_key(x[0])))
38 | 
39 | 
40 | _rescan_model_configs()  # initial populate of model config registry
41 | 
42 | 
43 | def list_models():
44 |     """enumerate available model architectures based on config files"""
45 |     return list(_MODEL_CONFIGS.keys())
46 | 
47 | 
48 | def add_model_config(path):
49 |     """add model config path or file and update registry"""
50 |     if not isinstance(path, Path):
51 |         path = Path(path)
52 |     _MODEL_CONFIG_PATHS.append(path)
53 |     _rescan_model_configs()
54 | 
55 | 
56 | def get_model_config(model_name):
57 |     if model_name in _MODEL_CONFIGS:
58 |         return deepcopy(_MODEL_CONFIGS[model_name])
59 |     else:
60 |         return None
61 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/apply_delta.py:
--------------------------------------------------------------------------------
 1 | """
 2 | Usage:
 3 | python3 -m fastchat.model.apply_delta --base ~/model_weights/llama-7b --target ~/model_weights/vicuna-7b --delta lmsys/vicuna-7b-delta
 4 | """
 5 | 
 6 | import argparse
 7 | 
 8 | import torch
 9 | from tqdm import tqdm
10 | from transformers import AutoTokenizer, AutoModelForCausalLM
11 | from llava import LlavaLlamaForCausalLM
12 | 
13 | 
14 | def apply_delta(base_model_path, target_model_path, delta_path):
15 |     print("Loading base model")
16 |     base = AutoModelForCausalLM.from_pretrained(base_model_path, torch_dtype=torch.float16, low_cpu_mem_usage=True)
17 | 
18 |     print("Loading delta")
19 |     delta = LlavaLlamaForCausalLM.from_pretrained(delta_path, torch_dtype=torch.float16, low_cpu_mem_usage=True)
20 |     delta_tokenizer = AutoTokenizer.from_pretrained(delta_path)
21 | 
22 |     print("Applying delta")
23 |     for name, param in tqdm(delta.state_dict().items(), desc="Applying delta"):
24 |         if name not in base.state_dict():
25 |             assert name in ["model.mm_projector.weight", "model.mm_projector.bias"], f"{name} not in base model"
26 |             continue
27 |         if param.data.shape == base.state_dict()[name].shape:
28 |             param.data += base.state_dict()[name]
29 |         else:
30 |             assert name in ["model.embed_tokens.weight", "lm_head.weight"], f"{name} dimension mismatch: {param.data.shape} vs {base.state_dict()[name].shape}"
31 |             bparam = base.state_dict()[name]
32 |             param.data[: bparam.shape[0], : bparam.shape[1]] += bparam
33 | 
34 |     print("Saving target model")
35 |     delta.save_pretrained(target_model_path)
36 |     delta_tokenizer.save_pretrained(target_model_path)
37 | 
38 | 
39 | if __name__ == "__main__":
40 |     parser = argparse.ArgumentParser()
41 |     parser.add_argument("--base-model-path", type=str, required=True)
42 |     parser.add_argument("--target-model-path", type=str, required=True)
43 |     parser.add_argument("--delta-path", type=str, required=True)
44 | 
45 |     args = parser.parse_args()
46 | 
47 |     apply_delta(args.base_model_path, args.target_model_path, args.delta_path)
48 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/groundingdino/models/GroundingDINO/csrc/MsDeformAttn/ms_deform_attn.h:
--------------------------------------------------------------------------------
 1 | /*!
 2 | **************************************************************************************************
 3 | * Deformable DETR
 4 | * Copyright (c) 2020 SenseTime. All Rights Reserved.
 5 | * Licensed under the Apache License, Version 2.0 [see LICENSE for details]
 6 | **************************************************************************************************
 7 | * Modified from https://github.com/chengdazhi/Deformable-Convolution-V2-PyTorch/tree/pytorch_1.0.0
 8 | **************************************************************************************************
 9 | */
10 | 
11 | #pragma once
12 | 
13 | #include "ms_deform_attn_cpu.h"
14 | 
15 | #ifdef WITH_CUDA
16 | #include "ms_deform_attn_cuda.h"
17 | #endif
18 | 
19 | namespace groundingdino {
20 | 
21 | at::Tensor
22 | ms_deform_attn_forward(
23 |     const at::Tensor &value, 
24 |     const at::Tensor &spatial_shapes,
25 |     const at::Tensor &level_start_index,
26 |     const at::Tensor &sampling_loc,
27 |     const at::Tensor &attn_weight,
28 |     const int im2col_step)
29 | {
30 |     if (value.type().is_cuda())
31 |     {
32 | #ifdef WITH_CUDA
33 |         return ms_deform_attn_cuda_forward(
34 |             value, spatial_shapes, level_start_index, sampling_loc, attn_weight, im2col_step);
35 | #else
36 |         AT_ERROR("Not compiled with GPU support");
37 | #endif
38 |     }
39 |     AT_ERROR("Not implemented on the CPU");
40 | }
41 | 
42 | std::vector<at::Tensor>
43 | ms_deform_attn_backward(
44 |     const at::Tensor &value, 
45 |     const at::Tensor &spatial_shapes,
46 |     const at::Tensor &level_start_index,
47 |     const at::Tensor &sampling_loc,
48 |     const at::Tensor &attn_weight,
49 |     const at::Tensor &grad_output,
50 |     const int im2col_step)
51 | {
52 |     if (value.type().is_cuda())
53 |     {
54 | #ifdef WITH_CUDA
55 |         return ms_deform_attn_cuda_backward(
56 |             value, spatial_shapes, level_start_index, sampling_loc, attn_weight, grad_output, im2col_step);
57 | #else
58 |         AT_ERROR("Not compiled with GPU support");
59 | #endif
60 |     }
61 |     AT_ERROR("Not implemented on the CPU");
62 | }
63 | 
64 | } // namespace groundingdino


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/serve/test_message.py:
--------------------------------------------------------------------------------
 1 | import argparse
 2 | import json
 3 | 
 4 | import requests
 5 | 
 6 | from llava.conversation import default_conversation
 7 | 
 8 | 
 9 | def main():
10 |     if args.worker_address:
11 |         worker_addr = args.worker_address
12 |     else:
13 |         controller_addr = args.controller_address
14 |         ret = requests.post(controller_addr + "/refresh_all_workers")
15 |         ret = requests.post(controller_addr + "/list_models")
16 |         models = ret.json()["models"]
17 |         models.sort()
18 |         print(f"Models: {models}")
19 | 
20 |         ret = requests.post(controller_addr + "/get_worker_address", json={"model": args.model_name})
21 |         worker_addr = ret.json()["address"]
22 |         print(f"worker_addr: {worker_addr}")
23 | 
24 |     if worker_addr == "":
25 |         return
26 | 
27 |     conv = default_conversation.copy()
28 |     conv.append_message(conv.roles[0], args.message)
29 |     prompt = conv.get_prompt()
30 | 
31 |     headers = {"User-Agent": "LLaVA Client"}
32 |     pload = {
33 |         "model": args.model_name,
34 |         "prompt": prompt,
35 |         "max_new_tokens": args.max_new_tokens,
36 |         "temperature": 0.7,
37 |         "stop": conv.sep,
38 |     }
39 |     response = requests.post(worker_addr + "/worker_generate_stream", headers=headers, json=pload, stream=True)
40 | 
41 |     print(prompt.replace(conv.sep, "\n"), end="")
42 |     for chunk in response.iter_lines(chunk_size=8192, decode_unicode=False, delimiter=b"\0"):
43 |         if chunk:
44 |             data = json.loads(chunk.decode("utf-8"))
45 |             output = data["text"].split(conv.sep)[-1]
46 |             print(output, end="\r")
47 |     print("")
48 | 
49 | 
50 | if __name__ == "__main__":
51 |     parser = argparse.ArgumentParser()
52 |     parser.add_argument("--controller-address", type=str, default="http://localhost:21001")
53 |     parser.add_argument("--worker-address", type=str)
54 |     parser.add_argument("--model-name", type=str, default="facebook/opt-350m")
55 |     parser.add_argument("--max-new-tokens", type=int, default=32)
56 |     parser.add_argument("--message", type=str, default="Tell me a story with more than 1000 words.")
57 |     args = parser.parse_args()
58 | 
59 |     main()
60 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/dev_eva_clip/eva_clip/hf_configs.py:
--------------------------------------------------------------------------------
 1 | # HF architecture dict:
 2 | arch_dict = {
 3 |     # https://huggingface.co/docs/transformers/model_doc/roberta#roberta
 4 |     "roberta": {
 5 |         "config_names": {
 6 |             "context_length": "max_position_embeddings",
 7 |             "vocab_size": "vocab_size",
 8 |             "width": "hidden_size",
 9 |             "heads": "num_attention_heads",
10 |             "layers": "num_hidden_layers",
11 |             "layer_attr": "layer",
12 |             "token_embeddings_attr": "embeddings",
13 |         },
14 |         "pooler": "mean_pooler",
15 |     },
16 |     # https://huggingface.co/docs/transformers/model_doc/xlm-roberta#transformers.XLMRobertaConfig
17 |     "xlm-roberta": {
18 |         "config_names": {
19 |             "context_length": "max_position_embeddings",
20 |             "vocab_size": "vocab_size",
21 |             "width": "hidden_size",
22 |             "heads": "num_attention_heads",
23 |             "layers": "num_hidden_layers",
24 |             "layer_attr": "layer",
25 |             "token_embeddings_attr": "embeddings",
26 |         },
27 |         "pooler": "mean_pooler",
28 |     },
29 |     # https://huggingface.co/docs/transformers/model_doc/mt5#mt5
30 |     "mt5": {
31 |         "config_names": {
32 |             # unlimited seqlen
33 |             # https://github.com/google-research/text-to-text-transfer-transformer/issues/273
34 |             # https://github.com/huggingface/transformers/blob/v4.24.0/src/transformers/models/t5/modeling_t5.py#L374
35 |             "context_length": "",
36 |             "vocab_size": "vocab_size",
37 |             "width": "d_model",
38 |             "heads": "num_heads",
39 |             "layers": "num_layers",
40 |             "layer_attr": "block",
41 |             "token_embeddings_attr": "embed_tokens",
42 |         },
43 |         "pooler": "mean_pooler",
44 |     },
45 |     "bert": {
46 |         "config_names": {
47 |             "context_length": "max_position_embeddings",
48 |             "vocab_size": "vocab_size",
49 |             "width": "hidden_size",
50 |             "heads": "num_attention_heads",
51 |             "layers": "num_hidden_layers",
52 |             "layer_attr": "layer",
53 |             "token_embeddings_attr": "embeddings",
54 |         },
55 |         "pooler": "mean_pooler",
56 |     },
57 | }
58 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/builder.py:
--------------------------------------------------------------------------------
 1 | import os
 2 | from .clip_encoder import CLIPVisionTower
 3 | from .imagebind import ImageBindWrapper
 4 | from .open_clip_encoder import OpenCLIPVisionTower
 5 | from .hf_vision import HFVisionTower
 6 | from .siglip_encoder import SigLipVisionTower
 7 | from .clip_encoder import CLIPVisionTower, CLIPVisionTowerS2
 8 | from .mlcd_encoder import MLCDVisionTower, MLCDVisionTowerS2
 9 | # from .eva_clip.eva_clip_encoder import EvaClipVisionTower
10 | # from .dev_eva_clip.eva_vit import EvaViTWrapper
11 | 
12 | 
13 | def build_vision_tower(vision_tower_cfg, **kwargs):
14 |     vision_tower = getattr(vision_tower_cfg, "mm_vision_tower", getattr(vision_tower_cfg, "vision_tower", None))
15 |     is_absolute_path_exists = os.path.exists(vision_tower)
16 |     use_s2 = getattr(vision_tower_cfg, "s2", False)
17 |     if is_absolute_path_exists or vision_tower.startswith("openai") or vision_tower.startswith("laion") or "ShareGPT4V" in vision_tower:
18 |         if use_s2:
19 |             return CLIPVisionTowerS2(vision_tower, args=vision_tower_cfg, **kwargs)
20 |         else:
21 |             return CLIPVisionTower(vision_tower, args=vision_tower_cfg, **kwargs)
22 |     elif "siglip" in vision_tower:
23 |         return SigLipVisionTower(vision_tower, vision_tower_cfg=vision_tower_cfg, **kwargs)
24 |     elif vision_tower.startswith("hf:"):
25 |         return HFVisionTower(vision_tower, args=vision_tower_cfg, **kwargs)
26 |     elif vision_tower in ["imagebind_huge"]:
27 |         return ImageBindWrapper(vision_tower, args=vision_tower_cfg, **kwargs)
28 |     elif vision_tower.startswith("open_clip_hub"):
29 |         return OpenCLIPVisionTower(vision_tower, args=vision_tower_cfg, **kwargs)
30 |     elif "mlcd-vit-bigG-patch14" in vision_tower:
31 |         if use_s2:
32 |             return MLCDVisionTowerS2(vision_tower, args=vision_tower_cfg, **kwargs)
33 |         else:
34 |             return MLCDVisionTower(vision_tower, args=vision_tower_cfg, **kwargs)
35 | 
36 |     # elif "internal-eva" in vision_tower.lower() or "eva02" in vision_tower.lower():
37 |     #     return EvaClipVisionTower(vision_tower, args=vision_tower_cfg, **kwargs)
38 |     # elif vision_tower in ["EVA-CLIP-8B", "EVA-CLIP-8B-plus"]:
39 |     #     return EvaViTWrapper(vision_tower, args=vision_tower_cfg, **kwargs)
40 | 
41 |     raise ValueError(f"Unknown vision tower: {vision_tower}")
42 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_projector/builder.py:
--------------------------------------------------------------------------------
 1 | import torch
 2 | import torch.nn as nn
 3 | import re
 4 | 
 5 | from .pooler_projector import PoolerProjector
 6 | 
 7 | 
 8 | class IdentityMap(nn.Module):
 9 |     def __init__(self):
10 |         super().__init__()
11 | 
12 |     def forward(self, x, *args, **kwargs):
13 |         return x
14 | 
15 |     @property
16 |     def config(self):
17 |         return {"mm_projector_type": "identity"}
18 | 
19 | 
20 | class SimpleResBlock(nn.Module):
21 |     def __init__(self, channels):
22 |         super().__init__()
23 |         self.pre_norm = nn.LayerNorm(channels)
24 | 
25 |         self.proj = nn.Sequential(nn.Linear(channels, channels), nn.GELU(), nn.Linear(channels, channels))
26 | 
27 |     def forward(self, x):
28 |         x = self.pre_norm(x)
29 |         return x + self.proj(x)
30 | 
31 | 
32 | def build_vision_projector(config, delay_load=False, **kwargs):
33 |     projector_type = getattr(config, "mm_projector_type", "linear")
34 | 
35 |     if projector_type == "linear":
36 |         return nn.Linear(config.mm_hidden_size, config.hidden_size)
37 | 
38 |     if projector_type == "pooler":
39 |         return PoolerProjector(config, kwargs["vision_cfg"])
40 | 
41 |     mlp_gelu_match = re.match(r"^mlp(\d+)x_gelu$", projector_type)
42 |     if mlp_gelu_match:
43 |         mlp_depth = int(mlp_gelu_match.group(1))
44 |         modules = [nn.Linear(config.mm_hidden_size, config.hidden_size)]
45 |         for _ in range(1, mlp_depth):
46 |             modules.append(nn.GELU())
47 |             modules.append(nn.Linear(config.hidden_size, config.hidden_size))
48 |         return nn.Sequential(*modules)
49 | 
50 |     mlp_gelu_resnet_match = re.match(r"^mlp(\d+)x_res(\d+)x_gelu$", projector_type)
51 |     if mlp_gelu_resnet_match:
52 |         mlp_depth = int(mlp_gelu_resnet_match.group(1))
53 |         res_depth = int(mlp_gelu_resnet_match.group(2))
54 |         modules = [nn.Linear(config.mm_hidden_size, config.hidden_size)]
55 |         for _ in range(1, mlp_depth):
56 |             modules.append(nn.GELU())
57 |             modules.append(nn.Linear(config.hidden_size, config.hidden_size))
58 |         for _ in range(res_depth):
59 |             modules.append(SimpleResBlock(config.hidden_size))
60 |         return nn.Sequential(*modules)
61 | 
62 |     if projector_type == "identity":
63 |         return IdentityMap()
64 | 
65 |     raise ValueError(f"Unknown projector type: {projector_type}")
66 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/groundingdino/models/registry.py:
--------------------------------------------------------------------------------
 1 | # ------------------------------------------------------------------------
 2 | # Grounding DINO
 3 | # url: https://github.com/IDEA-Research/GroundingDINO
 4 | # Copyright (c) 2023 IDEA. All Rights Reserved.
 5 | # Licensed under the Apache License, Version 2.0 [see LICENSE for details]
 6 | # ------------------------------------------------------------------------
 7 | # -*- coding: utf-8 -*-
 8 | # @Author: Yihao Chen
 9 | # @Date:   2021-08-16 16:03:17
10 | # @Last Modified by:   Shilong Liu
11 | # @Last Modified time: 2022-01-23 15:26
12 | # modified from mmcv
13 | 
14 | import inspect
15 | from functools import partial
16 | 
17 | 
18 | class Registry(object):
19 |     def __init__(self, name):
20 |         self._name = name
21 |         self._module_dict = dict()
22 | 
23 |     def __repr__(self):
24 |         format_str = self.__class__.__name__ + "(name={}, items={})".format(
25 |             self._name, list(self._module_dict.keys())
26 |         )
27 |         return format_str
28 | 
29 |     def __len__(self):
30 |         return len(self._module_dict)
31 | 
32 |     @property
33 |     def name(self):
34 |         return self._name
35 | 
36 |     @property
37 |     def module_dict(self):
38 |         return self._module_dict
39 | 
40 |     def get(self, key):
41 |         return self._module_dict.get(key, None)
42 | 
43 |     def registe_with_name(self, module_name=None, force=False):
44 |         return partial(self.register, module_name=module_name, force=force)
45 | 
46 |     def register(self, module_build_function, module_name=None, force=False):
47 |         """Register a module build function.
48 |         Args:
49 |             module (:obj:`nn.Module`): Module to be registered.
50 |         """
51 |         if not inspect.isfunction(module_build_function):
52 |             raise TypeError(
53 |                 "module_build_function must be a function, but got {}".format(
54 |                     type(module_build_function)
55 |                 )
56 |             )
57 |         if module_name is None:
58 |             module_name = module_build_function.__name__
59 |         if not force and module_name in self._module_dict:
60 |             raise KeyError("{} is already registered in {}".format(module_name, self.name))
61 |         self._module_dict[module_name] = module_build_function
62 | 
63 |         return module_build_function
64 | 
65 | 
66 | MODULE_BUILD_FUNCS = Registry("model build functions")
67 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/make_delta.py:
--------------------------------------------------------------------------------
 1 | """
 2 | Usage:
 3 | python3 -m llava.model.make_delta --base ~/model_weights/llama-7b --target ~/model_weights/llava-7b --delta ~/model_weights/llava-7b-delta --hub-repo-id liuhaotian/llava-7b-delta
 4 | """
 5 | 
 6 | import argparse
 7 | 
 8 | import torch
 9 | from tqdm import tqdm
10 | from transformers import AutoTokenizer, AutoModelForCausalLM
11 | from llava.model.utils import auto_upgrade
12 | 
13 | 
14 | def make_delta(base_model_path, target_model_path, delta_path, hub_repo_id):
15 |     print("Loading base model")
16 |     base = AutoModelForCausalLM.from_pretrained(base_model_path, torch_dtype=torch.float16, low_cpu_mem_usage=True)
17 | 
18 |     print("Loading target model")
19 |     auto_upgrade(target_model_path)
20 |     target = AutoModelForCausalLM.from_pretrained(target_model_path, torch_dtype=torch.float16, low_cpu_mem_usage=True)
21 | 
22 |     print("Calculating delta")
23 |     for name, param in tqdm(target.state_dict().items(), desc="Calculating delta"):
24 |         if name not in base.state_dict():
25 |             assert name in ["model.mm_projector.weight", "model.mm_projector.bias"], f"{name} not in base model"
26 |             continue
27 |         if param.data.shape == base.state_dict()[name].shape:
28 |             param.data -= base.state_dict()[name]
29 |         else:
30 |             assert name in ["model.embed_tokens.weight", "lm_head.weight"], f"{name} dimension mismatch: {param.data.shape} vs {base.state_dict()[name].shape}"
31 |             bparam = base.state_dict()[name]
32 |             param.data[: bparam.shape[0], : bparam.shape[1]] -= bparam
33 | 
34 |     print("Saving delta")
35 |     if hub_repo_id:
36 |         kwargs = {"push_to_hub": True, "repo_id": hub_repo_id}
37 |     else:
38 |         kwargs = {}
39 |     target.save_pretrained(delta_path, **kwargs)
40 |     target_tokenizer = AutoTokenizer.from_pretrained(target_model_path)
41 |     target_tokenizer.save_pretrained(delta_path, **kwargs)
42 | 
43 | 
44 | if __name__ == "__main__":
45 |     parser = argparse.ArgumentParser()
46 |     parser.add_argument("--base-model-path", type=str, required=True)
47 |     parser.add_argument("--target-model-path", type=str, required=True)
48 |     parser.add_argument("--delta-path", type=str, required=True)
49 |     parser.add_argument("--hub-repo-id", type=str, default=None)
50 |     args = parser.parse_args()
51 | 
52 |     make_delta(args.base_model_path, args.target_model_path, args.delta_path, args.hub_repo_id)
53 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/eva_clip/eva_clip_processors.py:
--------------------------------------------------------------------------------
 1 | """
 2 | # Adapted from https://github.com/baaivision/EVA/tree/master/EVA-CLIP
 3 | """
 4 | 
 5 | from torchvision import transforms
 6 | from torchvision.transforms.functional import InterpolationMode
 7 | from transformers.image_processing_utils import BatchFeature
 8 | from PIL import Image
 9 | from transformers.image_transforms import convert_to_rgb
10 | 
11 | 
12 | class BaseProcessor:
13 |     def __init__(self):
14 |         self.transform = lambda x: x
15 |         return
16 | 
17 |     def __call__(self, item):
18 |         return self.transform(item)
19 | 
20 | 
21 | class EvaClipImageBaseProcessor(BaseProcessor):
22 |     def __init__(self, mean=None, std=None):
23 |         self.mean = (0.48145466, 0.4578275, 0.40821073) if mean is None else mean
24 |         self.std = (0.26862954, 0.26130258, 0.27577711) if std is None else std
25 | 
26 |         self.normalize = transforms.Normalize(self.mean, self.std)
27 | 
28 |     @property
29 |     def image_mean(self):
30 |         return self.mean
31 | 
32 | 
33 | class EvaClipImageTrainProcessor(EvaClipImageBaseProcessor):
34 |     def __init__(self, image_size=224, mean=None, std=None, min_scale=0.5, max_scale=1.0):
35 |         super().__init__(mean=mean, std=std)
36 | 
37 |         self.transform = transforms.Compose(
38 |             [
39 |                 convert_to_rgb,
40 |                 transforms.Resize(
41 |                     image_size,
42 |                     interpolation=InterpolationMode.BICUBIC,
43 |                 ),
44 |                 transforms.CenterCrop(image_size),
45 |                 transforms.ToTensor(),
46 |                 self.normalize,
47 |             ]
48 |         )
49 | 
50 |         self.image_size = image_size
51 | 
52 |     def preprocess(self, images, return_tensors):
53 |         if isinstance(images, Image.Image):
54 |             images = [images]
55 |         else:
56 |             assert isinstance(images, list)
57 | 
58 |         transformed_images = [self.transform(image).numpy() for image in images]
59 |         data = {"pixel_values": transformed_images}
60 | 
61 |         return BatchFeature(data=data, tensor_type=return_tensors)
62 | 
63 |     def __call__(self, item):
64 |         return self.transform(item)
65 | 
66 |     @property
67 |     def crop_size(self):
68 |         return {"height": self.image_size, "width": self.image_size}
69 | 
70 |     @property
71 |     def size(self):
72 |         return {"shortest_edge": self.image_size}
73 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/.gitignore:
--------------------------------------------------------------------------------
  1 | # IDE
  2 | .idea/
  3 | .vscode/
  4 | 
  5 | # Byte-compiled / optimized / DLL files
  6 | __pycache__/
  7 | *.py[cod]
  8 | *$py.class
  9 | 
 10 | # C extensions
 11 | *.so
 12 | 
 13 | # Distribution / packaging
 14 | .Python
 15 | build/
 16 | develop-eggs/
 17 | dist/
 18 | downloads/
 19 | eggs/
 20 | .eggs/
 21 | lib/
 22 | lib64/
 23 | parts/
 24 | sdist/
 25 | var/
 26 | wheels/
 27 | pip-wheel-metadata/
 28 | share/python-wheels/
 29 | *.egg-info/
 30 | .installed.cfg
 31 | *.egg
 32 | MANIFEST
 33 | 
 34 | # PyInstaller
 35 | #  Usually these files are written by a python script from a template
 36 | #  before PyInstaller builds the exe, so as to inject date/other infos into it.
 37 | *.manifest
 38 | *.spec
 39 | 
 40 | # Installer logs
 41 | pip-log.txt
 42 | pip-delete-this-directory.txt
 43 | 
 44 | # Unit test / coverage reports
 45 | htmlcov/
 46 | .tox/
 47 | .nox/
 48 | .coverage
 49 | .coverage.*
 50 | .cache
 51 | nosetests.xml
 52 | coverage.xml
 53 | *.cover
 54 | *.py,cover
 55 | .hypothesis/
 56 | .pytest_cache/
 57 | 
 58 | # Translations
 59 | *.mo
 60 | *.pot
 61 | 
 62 | # Django stuff:
 63 | *.log
 64 | local_settings.py
 65 | db.sqlite3
 66 | db.sqlite3-journal
 67 | 
 68 | # Flask stuff:
 69 | instance/
 70 | .webassets-cache
 71 | 
 72 | # Scrapy stuff:
 73 | .scrapy
 74 | 
 75 | # Sphinx documentation
 76 | docs/_build/
 77 | 
 78 | # PyBuilder
 79 | target/
 80 | 
 81 | # Jupyter Notebook
 82 | .ipynb_checkpoints
 83 | 
 84 | # IPython
 85 | profile_default/
 86 | ipython_config.py
 87 | 
 88 | # pyenv
 89 | .python-version
 90 | 
 91 | # pipenv
 92 | #   According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
 93 | #   However, in case of collaboration, if having platform-specific dependencies or dependencies
 94 | #   having no cross-platform support, pipenv may install dependencies that don't work, or not
 95 | #   install all needed dependencies.
 96 | #Pipfile.lock
 97 | 
 98 | # PEP 582; used by e.g. github.com/David-OConnor/pyflow
 99 | __pypackages__/
100 | 
101 | # Celery stuff
102 | celerybeat-schedule
103 | celerybeat.pid
104 | 
105 | # SageMath parsed files
106 | *.sage.py
107 | 
108 | # Environments
109 | .env
110 | .venv
111 | env/
112 | venv/
113 | ENV/
114 | env.bak/
115 | venv.bak/
116 | 
117 | # Spyder project settings
118 | .spyderproject
119 | .spyproject
120 | 
121 | # Rope project settings
122 | .ropeproject
123 | 
124 | # mkdocs documentation
125 | /site
126 | 
127 | # mypy
128 | .mypy_cache/
129 | .dmypy.json
130 | dmypy.json
131 | 
132 | # Pyre type checker
133 | .pyre/
134 | 
135 | # vscode
136 | .vscode/
137 | output/
138 | outputs/
139 | subs/
140 | logs/
141 | 
142 | grounding/config/configs
143 | grounding/version.py
144 | 
145 | vis/
146 | tmp/


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/imagebind.py:
--------------------------------------------------------------------------------
 1 | import torch
 2 | import torch.nn as nn
 3 | 
 4 | from transformers import CLIPImageProcessor
 5 | 
 6 | try:
 7 |     from imagebind.models import imagebind_model
 8 |     from imagebind.models.imagebind_model import ModalityType
 9 |     from imagebind.data import load_and_transform_audio_data
10 | except ImportError:
11 |     pass
12 | 
13 | 
14 | class ImageBindWrapper(nn.Module):
15 |     def __init__(self, vision_tower, select_layer, select_feature="patch", delay_load=False):
16 |         super().__init__()
17 | 
18 |         self.is_loaded = False
19 | 
20 |         self.vision_tower_name = vision_tower
21 |         self.select_layer = select_layer
22 |         self.select_feature = select_feature
23 | 
24 |         if not delay_load:
25 |             self.load_model()
26 | 
27 |     def load_model(self):
28 |         self.image_processor = CLIPImageProcessor.from_pretrained("openai/clip-vit-large-patch14")
29 |         self.vision_tower = imagebind_model.imagebind_huge(pretrained=True)
30 |         for p in self.vision_tower.parameters():
31 |             p.requires_grad = False
32 |         self.vision_tower.eval()
33 |         self.is_loaded = True
34 | 
35 |     def train(self, mode=True):
36 |         self.training = mode
37 | 
38 |         if self.is_loaded:
39 |             self.vision_tower.eval()
40 | 
41 |     @torch.no_grad()
42 |     def forward(self, x):
43 |         if type(x) == dict:
44 |             if x["audios"] is not None:
45 |                 inputs = {ModalityType.AUDIO: load_and_transform_audio_data(x["audios"], device=self.device).half()}
46 |                 embeddings = self.vision_tower(inputs)
47 |                 audio_embedding = embeddings[ModalityType.AUDIO]
48 |                 return audio_embedding.unsqueeze(1)
49 |         else:
50 |             inputs = {ModalityType.VISION: x.to(dtype=self.dtype)}
51 |             embeddings = self.vision_tower(inputs)
52 |             vision_embedding = embeddings[ModalityType.VISION]
53 |             if vision_embedding.ndim == 2:
54 |                 return vision_embedding.unsqueeze(1)
55 |             if vision_embedding.shape[1] == 257:
56 |                 return vision_embedding[:, 1:]
57 |             raise ValueError(f"Unexpected shape: {vision_embedding.shape}")
58 | 
59 |     @property
60 |     def dummy_feature(self):
61 |         return torch.zeros(1, 1024, device=self.device, dtype=self.dtype)
62 | 
63 |     @property
64 |     def dtype(self):
65 |         return self.vision_tower.modality_preprocessors.vision.cls_token.dtype
66 | 
67 |     @property
68 |     def device(self):
69 |         return self.vision_tower.modality_preprocessors.vision.cls_token.device
70 | 
71 |     @property
72 |     def hidden_size(self):
73 |         return 1024
74 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/eva_clip/eva_clip_encoder.py:
--------------------------------------------------------------------------------
 1 | import torch
 2 | import torch.nn as nn
 3 | 
 4 | from .eva_clip_processors import EvaClipImageTrainProcessor
 5 | from .eva_vit import EVAEncoderWrapper
 6 | from .factory import list_models, add_model_config, get_model_config
 7 | 
 8 | from llava.utils import rank0_print
 9 | 
10 | 
11 | class EvaClipVisionTower(nn.Module):
12 |     def __init__(self, vision_tower, args, delay_load=False):
13 |         super().__init__()
14 | 
15 |         self.is_loaded = False
16 |         self.vision_tower_name = vision_tower
17 |         self.vision_tower_pretrained = args.vision_tower_pretrained
18 |         self.config = get_model_config(vision_tower)
19 | 
20 |         if not delay_load:
21 |             rank0_print(f"Loading EVA ViT: {self.vision_tower_name}")
22 |             self.load_model()
23 |         elif getattr(args, "unfreeze_mm_vision_tower", False):
24 |             # TODO: better detector is needed.
25 |             rank0_print(f"The checkpoint seems to contain `vision_tower` weights: `unfreeze_mm_vision_tower`: True.")
26 |             self.load_model()
27 |         elif hasattr(args, "mm_tunable_parts") and "mm_vision_tower" in args.mm_tunable_parts:
28 |             rank0_print(f"The checkpoint seems to contain `vision_tower` weights: `mm_tunable_parts` contains `mm_vision_tower`.")
29 |             self.load_model()
30 |         else:
31 |             self.cfg_only = self.config
32 | 
33 |     def load_model(self, device_map=None):
34 |         rank0_print(f"Pretrained: {self.vision_tower_pretrained}")
35 |         self.image_processor = EvaClipImageTrainProcessor(self.config["vision_cfg"]["image_size"])
36 |         self.vision_tower = EVAEncoderWrapper(self.vision_tower_pretrained, self.config)
37 |         rank0_print(f"Loaded image processor: {self.image_processor}")
38 |         self.vision_tower.requires_grad_(False)
39 |         self.is_loaded = True
40 | 
41 |     def forward(self, images):
42 |         if type(images) is list:
43 |             image_features = []
44 |             for image in images:
45 |                 image_feature = self.vision_tower(image.to(device=self.device, dtype=self.dtype).unsqueeze(0)).to(image.dtype)
46 |                 image_features.append(image_feature)
47 |         else:
48 |             image_features = self.vision_tower(images.to(device=self.device, dtype=self.dtype)).to(images.dtype)
49 | 
50 |         return image_features
51 | 
52 |     @property
53 |     def dtype(self):
54 |         return self.vision_tower.dtype
55 | 
56 |     @property
57 |     def device(self):
58 |         return self.vision_tower.device
59 | 
60 |     @property
61 |     def hidden_size(self):
62 |         return self.config["vision_cfg"]["width"]
63 | 
64 |     @property
65 |     def num_patches(self):
66 |         return (self.config["vision_cfg"]["image_size"] // self.config["vision_cfg"]["patch_size"]) ** 2
67 | 
68 |     @property
69 |     def num_patches_per_side(self):
70 |         return self.config["vision_cfg"]["image_size"] // self.config["vision_cfg"]["patch_size"]
71 | 
72 |     @property
73 |     def image_size(self):
74 |         return self.config["vision_cfg"]["image_size"]
75 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/reward_git.py:
--------------------------------------------------------------------------------
 1 | import os
 2 | import os
 3 | import torch
 4 | import requests
 5 | from PIL import Image
 6 | import numpy as np
 7 | import argparse
 8 | from transformers import AutoProcessor, AutoModelForCausalLM, AutoConfig, GitForCausalLM
 9 | 
10 | 
11 | class GIT:
12 |     def __init__(self, args):
13 | 
14 |         ckpt_path = args.git_ckpt_path
15 | 
16 |         self.processor = AutoProcessor.from_pretrained(ckpt_path)
17 |         config = AutoConfig.from_pretrained(ckpt_path)
18 |         self.model = GitForCausalLM(config)
19 |         # workaround for the zero3
20 |         ckpt = torch.load(os.path.join(ckpt_path, 'pytorch_model.bin'), map_location='cpu')
21 |         self.model.load_state_dict(ckpt, strict=False)
22 | 
23 |         # get yes and no token ids
24 |         self.yes_token_id = self.processor.tokenizer.encode('yes')[1] # [bos, yes, eos]
25 |         self.no_token_id = self.processor.tokenizer.encode('no')[1] # [bos, no, eos]
26 | 
27 |     @property
28 |     def __name__(self):
29 |         return 'GIT'
30 |     
31 |     def load_to_device(self, load_device):
32 | 
33 |         self.model.to(load_device)
34 | 
35 |         # freeze all parameters
36 |         for n, p in self.model.named_parameters():
37 |             p.requires_grad = False
38 |         self.model.eval()
39 | 
40 |             
41 |     def __call__(self, prompts, images, **kwargs):
42 |         device = list(self.model.parameters())[0].device
43 | 
44 |         # single generation
45 |         score = []
46 |         for i, (prompt, image) in enumerate(zip(prompts, images)):
47 |             # we do not calculate the score for spatial and numeracy tasks
48 |             if kwargs['task_type'][i] in ['spatial', 'numeracy']:
49 |                 score.append(0)
50 |                 continue
51 | 
52 |             # calculate attr nouns if exist, otherwise, calculate the nouns
53 |             if kwargs['attr_nouns'][i] is not None:
54 |                 key = 'attr_nouns'
55 |             else:
56 |                 key = 'nouns'
57 |                 if kwargs['nouns'][i] is None or len(kwargs['nouns'][i]) == 0:
58 |                     score.append(1)
59 |                     continue
60 | 
61 |             pixel_values = self.processor(images=image, return_tensors="pt").pixel_values.to(device)
62 |             temp_score = []
63 |             for idx, attr_noun in enumerate(kwargs[key][i]): # all the attr nouns should be the same, so we take the first one
64 |                 vqa_prompts = f"{attr_noun}?"
65 |                 input_ids = self.processor(text=vqa_prompts, add_special_tokens=False).input_ids
66 |                 input_ids = [self.processor.tokenizer.cls_token_id] + input_ids
67 |                 input_ids = torch.tensor(input_ids).unsqueeze(0).to(device)
68 | 
69 |                 logits = self.model(pixel_values=pixel_values, input_ids=input_ids, return_dict=True).logits[:, -1]
70 |                 probs = torch.softmax(logits, dim=1)
71 |                 prob_yes = probs[:, self.yes_token_id]
72 |                 prob_no = probs[:, self.no_token_id]
73 |                 temp_score.append((prob_yes / (prob_yes + prob_no)).cpu().numpy())
74 |             score.append(np.mean(temp_score).tolist())
75 | 
76 |         return score  # tensor
77 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_resampler/masked_drop.py:
--------------------------------------------------------------------------------
 1 | import torch
 2 | import torch.nn as nn
 3 | 
 4 | import random
 5 | 
 6 | 
 7 | class MaskedDrop(nn.Module):
 8 |     def __init__(self, model_args):
 9 |         super().__init__()
10 | 
11 |         self.mode = model_args.mm_mask_drop_mode
12 |         self.skip_percentage = model_args.mm_mask_drop_skip_percentage
13 |         self.ratio = model_args.mm_mask_drop_ratio
14 |         self.ratio_upper = model_args.mm_mask_drop_ratio_upper
15 |         self.ratio_lower = model_args.mm_mask_drop_ratio_lower
16 | 
17 |     def forward(self, image_features, *args, **kwargs):
18 | 
19 |         if not self.training:
20 |             return image_features
21 | 
22 |         if self.skip_percentage > random.random():
23 |             return image_features
24 | 
25 |         masked_features = []
26 | 
27 |         for image_feature in image_features:
28 |             num_tokens = image_feature.shape[0]
29 |             if self.mode == "fixed":
30 |                 num_keep = int(num_tokens * self.ratio)
31 |                 masked_features.append(self.random_masking(image_feature.unsqueeze(0), num_keep)[0][0])
32 |             elif self.mode == "range":
33 |                 num_keep = int(num_tokens * random.uniform(self.ratio_lower, self.ratio_upper))
34 |                 masked_features.append(self.random_masking(image_feature.unsqueeze(0), num_keep)[0])
35 |             elif self.mode == "cls_only":
36 |                 masked_features.append(image_feature[0:1])
37 |             else:
38 |                 raise ValueError(f"Unexpected masked drop mode: {self.mode}")
39 | 
40 |         if self.mode not in ["range"] and (type(image_features) is not list or self.mode in ["cls_only"]):
41 |             masked_features = torch.stack(masked_features, dim=0)
42 | 
43 |         return masked_features
44 | 
45 |     @property
46 |     def config(self):
47 |         return {
48 |             "mm_resampler_type": "masked_drop",
49 |             "mm_mask_drop_mode": self.mode,
50 |             "mm_mask_drop_skip_percentage": self.skip_percentage,
51 |             "mm_mask_drop_ratio": self.ratio,
52 |             "mm_mask_drop_ratio_upper": self.ratio_upper,
53 |             "mm_mask_drop_ratio_lower": self.ratio_lower,
54 |         }
55 | 
56 |     def random_masking(self, x, len_keep):
57 |         """
58 |         Perform per-sample random masking by per-sample shuffling.
59 |         Per-sample shuffling is done by argsort random noise.
60 |         x: [N, L, D], sequence
61 |         """
62 |         N, L, D = x.shape  # batch, length, dim
63 | 
64 |         noise = torch.rand(N, L, device=x.device)  # noise in [0, 1]
65 | 
66 |         # sort noise for each sample
67 |         ids_shuffle = torch.argsort(noise, dim=1)  # ascend: small is keep, large is remove
68 |         ids_restore = torch.argsort(ids_shuffle, dim=1)
69 | 
70 |         # keep the first subset
71 |         ids_keep = ids_shuffle[:, :len_keep]
72 |         x_masked = torch.gather(x, dim=1, index=ids_keep.unsqueeze(-1).repeat(1, 1, D))
73 | 
74 |         # generate the binary mask: 0 is keep, 1 is remove
75 |         mask = torch.ones([N, L], device=x.device)
76 |         mask[:, :len_keep] = 0
77 |         # unshuffle to get the binary mask
78 |         mask = torch.gather(mask, dim=1, index=ids_restore)
79 | 
80 |         return x_masked, mask, ids_restore
81 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/janus/utils/io.py:
--------------------------------------------------------------------------------
 1 | # Copyright (c) 2023-2024 DeepSeek.
 2 | #
 3 | # Permission is hereby granted, free of charge, to any person obtaining a copy of
 4 | # this software and associated documentation files (the "Software"), to deal in
 5 | # the Software without restriction, including without limitation the rights to
 6 | # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
 7 | # the Software, and to permit persons to whom the Software is furnished to do so,
 8 | # subject to the following conditions:
 9 | #
10 | # The above copyright notice and this permission notice shall be included in all
11 | # copies or substantial portions of the Software.
12 | #
13 | # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
14 | # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
15 | # FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
16 | # COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
17 | # IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
18 | # CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
19 | 
20 | import json
21 | from typing import Dict, List
22 | 
23 | import PIL.Image
24 | import torch
25 | import base64
26 | import io
27 | from transformers import AutoModelForCausalLM
28 | 
29 | from janus.models import MultiModalityCausalLM, VLChatProcessor
30 | 
31 | 
32 | def load_pretrained_model(model_path: str):
33 |     vl_chat_processor: VLChatProcessor = VLChatProcessor.from_pretrained(model_path)
34 |     tokenizer = vl_chat_processor.tokenizer
35 | 
36 |     vl_gpt: MultiModalityCausalLM = AutoModelForCausalLM.from_pretrained(
37 |         model_path, trust_remote_code=True
38 |     )
39 |     vl_gpt = vl_gpt.to(torch.bfloat16).cuda().eval()
40 | 
41 |     return tokenizer, vl_chat_processor, vl_gpt
42 | 
43 | 
44 | def load_pil_images(conversations: List[Dict[str, str]]) -> List[PIL.Image.Image]:
45 |     """
46 | 
47 |     Support file path or base64 images.
48 | 
49 |     Args:
50 |         conversations (List[Dict[str, str]]): the conversations with a list of messages. An example is :
51 |             [
52 |                 {
53 |                     "role": "User",
54 |                     "content": "<image_placeholder>\nExtract all information from this image and convert them into markdown format.",
55 |                     "images": ["./examples/table_datasets.png"]
56 |                 },
57 |                 {"role": "Assistant", "content": ""},
58 |             ]
59 | 
60 |     Returns:
61 |         pil_images (List[PIL.Image.Image]): the list of PIL images.
62 | 
63 |     """
64 | 
65 |     pil_images = []
66 | 
67 |     for message in conversations:
68 |         if "images" not in message:
69 |             continue
70 | 
71 |         for image_data in message["images"]:
72 |             if image_data.startswith("data:image"):
73 |                 # Image data is in base64 format
74 |                 _, image_data = image_data.split(",", 1)
75 |                 image_bytes = base64.b64decode(image_data)
76 |                 pil_img = PIL.Image.open(io.BytesIO(image_bytes))
77 |             else:
78 |                 # Image data is a file path
79 |                 pil_img = PIL.Image.open(image_data)
80 |             pil_img = pil_img.convert("RGB")
81 |             pil_images.append(pil_img)
82 | 
83 |     return pil_images
84 | 
85 | 
86 | def load_json(filepath):
87 |     with open(filepath, "r") as f:
88 |         data = json.load(f)
89 |         return data
90 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/groundingdino/util/logger.py:
--------------------------------------------------------------------------------
 1 | # Copyright (c) Facebook, Inc. and its affiliates. All Rights Reserved
 2 | import functools
 3 | import logging
 4 | import os
 5 | import sys
 6 | 
 7 | from termcolor import colored
 8 | 
 9 | 
10 | class _ColorfulFormatter(logging.Formatter):
11 |     def __init__(self, *args, **kwargs):
12 |         self._root_name = kwargs.pop("root_name") + "."
13 |         self._abbrev_name = kwargs.pop("abbrev_name", "")
14 |         if len(self._abbrev_name):
15 |             self._abbrev_name = self._abbrev_name + "."
16 |         super(_ColorfulFormatter, self).__init__(*args, **kwargs)
17 | 
18 |     def formatMessage(self, record):
19 |         record.name = record.name.replace(self._root_name, self._abbrev_name)
20 |         log = super(_ColorfulFormatter, self).formatMessage(record)
21 |         if record.levelno == logging.WARNING:
22 |             prefix = colored("WARNING", "red", attrs=["blink"])
23 |         elif record.levelno == logging.ERROR or record.levelno == logging.CRITICAL:
24 |             prefix = colored("ERROR", "red", attrs=["blink", "underline"])
25 |         else:
26 |             return log
27 |         return prefix + " " + log
28 | 
29 | 
30 | # so that calling setup_logger multiple times won't add many handlers
31 | @functools.lru_cache()
32 | def setup_logger(output=None, distributed_rank=0, *, color=True, name="imagenet", abbrev_name=None):
33 |     """
34 |     Initialize the detectron2 logger and set its verbosity level to "INFO".
35 | 
36 |     Args:
37 |         output (str): a file name or a directory to save log. If None, will not save log file.
38 |             If ends with ".txt" or ".log", assumed to be a file name.
39 |             Otherwise, logs will be saved to `output/log.txt`.
40 |         name (str): the root module name of this logger
41 | 
42 |     Returns:
43 |         logging.Logger: a logger
44 |     """
45 |     logger = logging.getLogger(name)
46 |     logger.setLevel(logging.DEBUG)
47 |     logger.propagate = False
48 | 
49 |     if abbrev_name is None:
50 |         abbrev_name = name
51 | 
52 |     plain_formatter = logging.Formatter(
53 |         "[%(asctime)s.%(msecs)03d]: %(message)s", datefmt="%m/%d %H:%M:%S"
54 |     )
55 |     # stdout logging: master only
56 |     if distributed_rank == 0:
57 |         ch = logging.StreamHandler(stream=sys.stdout)
58 |         ch.setLevel(logging.DEBUG)
59 |         if color:
60 |             formatter = _ColorfulFormatter(
61 |                 colored("[%(asctime)s.%(msecs)03d]: ", "green") + "%(message)s",
62 |                 datefmt="%m/%d %H:%M:%S",
63 |                 root_name=name,
64 |                 abbrev_name=str(abbrev_name),
65 |             )
66 |         else:
67 |             formatter = plain_formatter
68 |         ch.setFormatter(formatter)
69 |         logger.addHandler(ch)
70 | 
71 |     # file logging: all workers
72 |     if output is not None:
73 |         if output.endswith(".txt") or output.endswith(".log"):
74 |             filename = output
75 |         else:
76 |             filename = os.path.join(output, "log.txt")
77 |         if distributed_rank > 0:
78 |             filename = filename + f".rank{distributed_rank}"
79 |         os.makedirs(os.path.dirname(filename), exist_ok=True)
80 | 
81 |         fh = logging.StreamHandler(_cached_log_stream(filename))
82 |         fh.setLevel(logging.DEBUG)
83 |         fh.setFormatter(plain_formatter)
84 |         logger.addHandler(fh)
85 | 
86 |     return logger
87 | 
88 | 
89 | # cache the opened file object, so that different calls to `setup_logger`
90 | # with the same file name can safely write to the same file.
91 | @functools.lru_cache(maxsize=None)
92 | def _cached_log_stream(filename):
93 |     return open(filename, "a")
94 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/test.ipynb:
--------------------------------------------------------------------------------
  1 | {
  2 |  "cells": [
  3 |   {
  4 |    "cell_type": "code",
  5 |    "execution_count": 2,
  6 |    "metadata": {},
  7 |    "outputs": [
  8 |     {
  9 |      "name": "stdout",
 10 |      "output_type": "stream",
 11 |      "text": [
 12 |       "final text_encoder_type: bert-base-uncased\n"
 13 |      ]
 14 |     },
 15 |     {
 16 |      "data": {
 17 |       "application/json": {
 18 |        "ascii": false,
 19 |        "bar_format": null,
 20 |        "colour": null,
 21 |        "elapsed": 0.014210224151611328,
 22 |        "initial": 0,
 23 |        "n": 0,
 24 |        "ncols": null,
 25 |        "nrows": null,
 26 |        "postfix": null,
 27 |        "prefix": "Downloading model.safetensors",
 28 |        "rate": null,
 29 |        "total": 440449768,
 30 |        "unit": "B",
 31 |        "unit_divisor": 1000,
 32 |        "unit_scale": true
 33 |       },
 34 |       "application/vnd.jupyter.widget-view+json": {
 35 |        "model_id": "5922f34578364d36afa13de9f01254bd",
 36 |        "version_major": 2,
 37 |        "version_minor": 0
 38 |       },
 39 |       "text/plain": [
 40 |        "Downloading model.safetensors:   0%|          | 0.00/440M [00:00<?, ?B/s]"
 41 |       ]
 42 |      },
 43 |      "metadata": {},
 44 |      "output_type": "display_data"
 45 |     },
 46 |     {
 47 |      "name": "stderr",
 48 |      "output_type": "stream",
 49 |      "text": [
 50 |       "/root/miniconda3/lib/python3.8/site-packages/transformers/modeling_utils.py:881: FutureWarning: The `device` argument is deprecated and will be removed in v5 of Transformers.\n",
 51 |       "  warnings.warn(\n",
 52 |       "/root/miniconda3/lib/python3.8/site-packages/torch/utils/checkpoint.py:31: UserWarning: None of the inputs have requires_grad=True. Gradients will be None\n",
 53 |       "  warnings.warn(\"None of the inputs have requires_grad=True. Gradients will be None\")\n"
 54 |      ]
 55 |     },
 56 |     {
 57 |      "data": {
 58 |       "text/plain": [
 59 |        "True"
 60 |       ]
 61 |      },
 62 |      "execution_count": 2,
 63 |      "metadata": {},
 64 |      "output_type": "execute_result"
 65 |     }
 66 |    ],
 67 |    "source": [
 68 |     "from groundingdino.util.inference import load_model, load_image, predict, annotate\n",
 69 |     "import cv2\n",
 70 |     "\n",
 71 |     "model = load_model(\"groundingdino/config/GroundingDINO_SwinT_OGC.py\", \"../04-06-segment-anything/weights/groundingdino_swint_ogc.pth\")\n",
 72 |     "IMAGE_PATH = \".asset/cat_dog.jpeg\"\n",
 73 |     "TEXT_PROMPT = \"chair . person . dog .\"\n",
 74 |     "BOX_TRESHOLD = 0.35\n",
 75 |     "TEXT_TRESHOLD = 0.25\n",
 76 |     "\n",
 77 |     "image_source, image = load_image(IMAGE_PATH)\n",
 78 |     "\n",
 79 |     "boxes, logits, phrases = predict(\n",
 80 |     "    model=model,\n",
 81 |     "    image=image,\n",
 82 |     "    caption=TEXT_PROMPT,\n",
 83 |     "    box_threshold=BOX_TRESHOLD,\n",
 84 |     "    text_threshold=TEXT_TRESHOLD\n",
 85 |     ")\n",
 86 |     "\n",
 87 |     "annotated_frame = annotate(image_source=image_source, boxes=boxes, logits=logits, phrases=phrases)\n",
 88 |     "cv2.imwrite(\"annotated_image.jpg\", annotated_frame)"
 89 |    ]
 90 |   }
 91 |  ],
 92 |  "metadata": {
 93 |   "kernelspec": {
 94 |    "display_name": "base",
 95 |    "language": "python",
 96 |    "name": "python3"
 97 |   },
 98 |   "language_info": {
 99 |    "codemirror_mode": {
100 |     "name": "ipython",
101 |     "version": 3
102 |    },
103 |    "file_extension": ".py",
104 |    "mimetype": "text/x-python",
105 |    "name": "python",
106 |    "nbconvert_exporter": "python",
107 |    "pygments_lexer": "ipython3",
108 |    "version": "3.8.10"
109 |   },
110 |   "orig_nbformat": 4
111 |  },
112 |  "nbformat": 4,
113 |  "nbformat_minor": 2
114 | }
115 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/train/llava_trainer_eval.py:
--------------------------------------------------------------------------------
 1 | import json
 2 | import subprocess
 3 | 
 4 | from llava.train.llava_trainer import LLaVATrainer
 5 | 
 6 | 
 7 | class LLaVAEvalTrainer(LLaVATrainer):
 8 |     def evaluate(self, evaluate_args):
 9 |         cmd = f"accelerate launch --num_processes {evaluate_args.eval_num_processes} -m lmms_eval \
10 |                 --model {evaluate_args.model} \
11 |                 --model_args {evaluate_args.model_args} \
12 |                 --tasks {evaluate_args.task_names} \
13 |                 --batch_size {evaluate_args.batch_size} \
14 |                 --log_samples_suffix {evaluate_args.log_samples_suffix} \
15 |                 --output_path {evaluate_args.output_path}"
16 |         if evaluate_args.limit:
17 |             cmd += f" --limit {evaluate_args.limit}"
18 |         if evaluate_args.num_fewshot:
19 |             cmd += f" --num_fewshot {evaluate_args.num_fewshot}"
20 |         if evaluate_args.gen_kwargs != "":
21 |             cmd += f" --gen_kwargs {evaluate_args.gen_kwargs}"
22 |         if evaluate_args.log_samples:
23 |             cmd += f" --log_samples"
24 |         else:
25 |             assert False, "Please log samples so that the result can be parsed"
26 |         results = subprocess.run([cmd], shell=True, capture_output=True, text=True)
27 |         try:
28 |             result_file_index_start = results.stdout.index("Saved samples to ")
29 |             result_file_index_end = results.stdout.index(f".json")
30 |             result_file_index_start += len("Saved samples to ")
31 |             file = results.stdout[result_file_index_start:result_file_index_end]
32 |         except:
33 |             result_file_index_start = results.stderr.index("Saved samples to ")
34 |             result_file_index_end = results.stderr.index(f".json")
35 |             result_file_index_start += len("Saved samples to ")
36 |             file = results.stderr[result_file_index_start:result_file_index_end]
37 |         file = file.split("/")[:-1]
38 |         file = "/".join(file) + "/results.json"
39 |         with open(file, "r") as f:
40 |             lmms_eval_results = json.load(f)
41 |         result_dict = {}
42 |         tasks_list = evaluate_args.task_names.split(",")
43 |         for task in tasks_list:
44 |             task_results = lmms_eval_results["results"][task]
45 |             for k, v in task_results.items():
46 |                 if k != "alias" and "stderr" not in k:
47 |                     metric = k.split(",")[0]
48 |                     result_dict[f"{task}_{metric}"] = v
49 |         return result_dict
50 | 
51 |     """def evaluate(self, evaluate_args):
52 |         initialize_tasks()
53 |         tasks_list = evaluate_args.task_names.split(",")
54 |         result_dict = {}
55 |         results = evaluator.simple_evaluate(
56 |             model=evaluate_args.model,
57 |             model_args=evaluate_args.model_args,
58 |             tasks=tasks_list,
59 |             num_fewshot=evaluate_args.num_fewshot,
60 |             batch_size=evaluate_args.batch_size,
61 |             device=evaluate_args.device,
62 |             limit=evaluate_args.limit,
63 |             check_integrity=evaluate_args.check_integrity,
64 |             show_task_to_terminal=evaluate_args.show_task_to_terminal,
65 |             log_samples=evaluate_args.log_samples,
66 |             gen_kwargs=evaluate_args.gen_kwargs,
67 |             cli_args=evaluate_args,
68 |         )
69 |         for task in tasks_list:
70 |             task_results = results["results"][task]
71 |             for k,v in task_results.items():
72 |                 if k != "alias" and "stderr" not in k:
73 |                     metric = k.split(",")[0]
74 |                     result_dict[f"{task}_{metric}"] = v
75 |             
76 |         return result_dict"""
77 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/dev_eva_clip/eva_clip/transform.py:
--------------------------------------------------------------------------------
  1 | from typing import Optional, Sequence, Tuple
  2 | 
  3 | import torch
  4 | import torch.nn as nn
  5 | import torchvision.transforms.functional as F
  6 | 
  7 | from torchvision.transforms import Normalize, Compose, RandomResizedCrop, InterpolationMode, ToTensor, Resize, CenterCrop
  8 | 
  9 | from .constants import OPENAI_DATASET_MEAN, OPENAI_DATASET_STD
 10 | 
 11 | 
 12 | class ResizeMaxSize(nn.Module):
 13 | 
 14 |     def __init__(self, max_size, interpolation=InterpolationMode.BICUBIC, fn="max", fill=0):
 15 |         super().__init__()
 16 |         if not isinstance(max_size, int):
 17 |             raise TypeError(f"Size should be int. Got {type(max_size)}")
 18 |         self.max_size = max_size
 19 |         self.interpolation = interpolation
 20 |         self.fn = min if fn == "min" else min
 21 |         self.fill = fill
 22 | 
 23 |     def forward(self, img):
 24 |         if isinstance(img, torch.Tensor):
 25 |             height, width = img.shape[:2]
 26 |         else:
 27 |             width, height = img.size
 28 |         scale = self.max_size / float(max(height, width))
 29 |         if scale != 1.0:
 30 |             new_size = tuple(round(dim * scale) for dim in (height, width))
 31 |             img = F.resize(img, new_size, self.interpolation)
 32 |             pad_h = self.max_size - new_size[0]
 33 |             pad_w = self.max_size - new_size[1]
 34 |             img = F.pad(img, padding=[pad_w // 2, pad_h // 2, pad_w - pad_w // 2, pad_h - pad_h // 2], fill=self.fill)
 35 |         return img
 36 | 
 37 | 
 38 | def _convert_to_rgb(image):
 39 |     return image.convert("RGB")
 40 | 
 41 | 
 42 | # class CatGen(nn.Module):
 43 | #     def __init__(self, num=4):
 44 | #         self.num = num
 45 | #     def mixgen_batch(image, text):
 46 | #         batch_size = image.shape[0]
 47 | #         index = np.random.permutation(batch_size)
 48 | 
 49 | #         cat_images = []
 50 | #         for i in range(batch_size):
 51 | #             # image mixup
 52 | #             image[i,:] = lam * image[i,:] + (1 - lam) * image[index[i],:]
 53 | #             # text concat
 54 | #             text[i] = tokenizer((str(text[i]) + " " + str(text[index[i]])))[0]
 55 | #         text = torch.stack(text)
 56 | #         return image, text
 57 | 
 58 | 
 59 | def image_transform(
 60 |     image_size: int,
 61 |     is_train: bool,
 62 |     mean: Optional[Tuple[float, ...]] = None,
 63 |     std: Optional[Tuple[float, ...]] = None,
 64 |     resize_longest_max: bool = False,
 65 |     fill_color: int = 0,
 66 | ):
 67 |     mean = mean or OPENAI_DATASET_MEAN
 68 |     if not isinstance(mean, (list, tuple)):
 69 |         mean = (mean,) * 3
 70 | 
 71 |     std = std or OPENAI_DATASET_STD
 72 |     if not isinstance(std, (list, tuple)):
 73 |         std = (std,) * 3
 74 | 
 75 |     if isinstance(image_size, (list, tuple)) and image_size[0] == image_size[1]:
 76 |         # for square size, pass size as int so that Resize() uses aspect preserving shortest edge
 77 |         image_size = image_size[0]
 78 | 
 79 |     normalize = Normalize(mean=mean, std=std)
 80 |     if is_train:
 81 |         return Compose(
 82 |             [
 83 |                 RandomResizedCrop(image_size, scale=(0.9, 1.0), interpolation=InterpolationMode.BICUBIC),
 84 |                 _convert_to_rgb,
 85 |                 ToTensor(),
 86 |                 normalize,
 87 |             ]
 88 |         )
 89 |     else:
 90 |         if resize_longest_max:
 91 |             transforms = [ResizeMaxSize(image_size, fill=fill_color)]
 92 |         else:
 93 |             transforms = [
 94 |                 Resize(image_size, interpolation=InterpolationMode.BICUBIC),
 95 |                 CenterCrop(image_size),
 96 |             ]
 97 |         transforms.extend(
 98 |             [
 99 |                 _convert_to_rgb,
100 |                 ToTensor(),
101 |                 normalize,
102 |             ]
103 |         )
104 |         return Compose(transforms)
105 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/groundingdino/util/vl_utils.py:
--------------------------------------------------------------------------------
  1 | import os
  2 | import random
  3 | from typing import List
  4 | 
  5 | import torch
  6 | 
  7 | 
  8 | def create_positive_map_from_span(tokenized, token_span, max_text_len=256):
  9 |     """construct a map such that positive_map[i,j] = True iff box i is associated to token j
 10 |     Input:
 11 |         - tokenized:
 12 |             - input_ids: Tensor[1, ntokens]
 13 |             - attention_mask: Tensor[1, ntokens]
 14 |         - token_span: list with length num_boxes.
 15 |             - each item: [start_idx, end_idx]
 16 |     """
 17 |     positive_map = torch.zeros((len(token_span), max_text_len), dtype=torch.float)
 18 |     for j, tok_list in enumerate(token_span):
 19 |         for (beg, end) in tok_list:
 20 |             beg_pos = tokenized.char_to_token(beg)
 21 |             end_pos = tokenized.char_to_token(end - 1)
 22 |             if beg_pos is None:
 23 |                 try:
 24 |                     beg_pos = tokenized.char_to_token(beg + 1)
 25 |                     if beg_pos is None:
 26 |                         beg_pos = tokenized.char_to_token(beg + 2)
 27 |                 except:
 28 |                     beg_pos = None
 29 |             if end_pos is None:
 30 |                 try:
 31 |                     end_pos = tokenized.char_to_token(end - 2)
 32 |                     if end_pos is None:
 33 |                         end_pos = tokenized.char_to_token(end - 3)
 34 |                 except:
 35 |                     end_pos = None
 36 |             if beg_pos is None or end_pos is None:
 37 |                 continue
 38 | 
 39 |             assert beg_pos is not None and end_pos is not None
 40 |             if os.environ.get("SHILONG_DEBUG_ONLY_ONE_POS", None) == "TRUE":
 41 |                 positive_map[j, beg_pos] = 1
 42 |                 break
 43 |             else:
 44 |                 positive_map[j, beg_pos : end_pos + 1].fill_(1)
 45 | 
 46 |     return positive_map / (positive_map.sum(-1)[:, None] + 1e-6)
 47 | 
 48 | 
 49 | def build_captions_and_token_span(cat_list, force_lowercase):
 50 |     """
 51 |     Return:
 52 |         captions: str
 53 |         cat2tokenspan: dict
 54 |             {
 55 |                 'dog': [[0, 2]],
 56 |                 ...
 57 |             }
 58 |     """
 59 | 
 60 |     cat2tokenspan = {}
 61 |     captions = ""
 62 |     for catname in cat_list:
 63 |         class_name = catname
 64 |         if force_lowercase:
 65 |             class_name = class_name.lower()
 66 |         if "/" in class_name:
 67 |             class_name_list: List = class_name.strip().split("/")
 68 |             class_name_list.append(class_name)
 69 |             class_name: str = random.choice(class_name_list)
 70 | 
 71 |         tokens_positive_i = []
 72 |         subnamelist = [i.strip() for i in class_name.strip().split(" ")]
 73 |         for subname in subnamelist:
 74 |             if len(subname) == 0:
 75 |                 continue
 76 |             if len(captions) > 0:
 77 |                 captions = captions + " "
 78 |             strat_idx = len(captions)
 79 |             end_idx = strat_idx + len(subname)
 80 |             tokens_positive_i.append([strat_idx, end_idx])
 81 |             captions = captions + subname
 82 | 
 83 |         if len(tokens_positive_i) > 0:
 84 |             captions = captions + " ."
 85 |             cat2tokenspan[class_name] = tokens_positive_i
 86 | 
 87 |     return captions, cat2tokenspan
 88 | 
 89 | 
 90 | def build_id2posspan_and_caption(category_dict: dict):
 91 |     """Build id2pos_span and caption from category_dict
 92 | 
 93 |     Args:
 94 |         category_dict (dict): category_dict
 95 |     """
 96 |     cat_list = [item["name"].lower() for item in category_dict]
 97 |     id2catname = {item["id"]: item["name"].lower() for item in category_dict}
 98 |     caption, cat2posspan = build_captions_and_token_span(cat_list, force_lowercase=True)
 99 |     id2posspan = {catid: cat2posspan[catname] for catid, catname in id2catname.items()}
100 |     return id2posspan, caption
101 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/janus/models/projector.py:
--------------------------------------------------------------------------------
  1 | # Copyright (c) 2023-2024 DeepSeek.
  2 | #
  3 | # Permission is hereby granted, free of charge, to any person obtaining a copy of
  4 | # this software and associated documentation files (the "Software"), to deal in
  5 | # the Software without restriction, including without limitation the rights to
  6 | # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
  7 | # the Software, and to permit persons to whom the Software is furnished to do so,
  8 | # subject to the following conditions:
  9 | #
 10 | # The above copyright notice and this permission notice shall be included in all
 11 | # copies or substantial portions of the Software.
 12 | #
 13 | # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
 14 | # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
 15 | # FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
 16 | # COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
 17 | # IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
 18 | # CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
 19 | 
 20 | from typing import Tuple, Union
 21 | 
 22 | import torch
 23 | import torch.nn as nn
 24 | from attrdict import AttrDict
 25 | 
 26 | 
 27 | class MlpProjector(nn.Module):
 28 |     def __init__(self, cfg):
 29 |         super().__init__()
 30 | 
 31 |         self.cfg = cfg
 32 | 
 33 |         if cfg.projector_type == "identity":
 34 |             modules = nn.Identity()
 35 | 
 36 |         elif cfg.projector_type == "linear":
 37 |             modules = nn.Linear(cfg.input_dim, cfg.n_embed)
 38 | 
 39 |         elif cfg.projector_type == "mlp_gelu":
 40 |             mlp_depth = cfg.get("depth", 1)
 41 |             modules = [nn.Linear(cfg.input_dim, cfg.n_embed)]
 42 |             for _ in range(1, mlp_depth):
 43 |                 modules.append(nn.GELU())
 44 |                 modules.append(nn.Linear(cfg.n_embed, cfg.n_embed))
 45 |             modules = nn.Sequential(*modules)
 46 | 
 47 |         elif cfg.projector_type == "low_high_hybrid_split_mlp_gelu":
 48 |             mlp_depth = cfg.get("depth", 1)
 49 |             self.high_up_proj = nn.Linear(cfg.input_dim, cfg.n_embed // 2)
 50 |             self.low_up_proj = nn.Linear(cfg.input_dim, cfg.n_embed // 2)
 51 | 
 52 |             modules = []
 53 |             for _ in range(1, mlp_depth):
 54 |                 modules.append(nn.GELU())
 55 |                 modules.append(nn.Linear(cfg.n_embed, cfg.n_embed))
 56 |             modules = nn.Sequential(*modules)
 57 | 
 58 |         else:
 59 |             raise ValueError(f"Unknown projector type: {cfg.projector_type}")
 60 | 
 61 |         self.layers = modules
 62 | 
 63 |     def forward(
 64 |         self, x_or_tuple: Union[Tuple[torch.Tensor, torch.Tensor], torch.Tensor]
 65 |     ):
 66 |         """
 67 | 
 68 |         Args:
 69 |             x_or_tuple (Union[Tuple[torch.Tensor, torch.Tensor], torch.Tensor]:  if it is a tuple of torch.Tensor,
 70 |                 then it comes from the hybrid vision encoder, and x = high_res_x, low_res_x);
 71 |                 otherwise it is the feature from the single vision encoder.
 72 | 
 73 |         Returns:
 74 |             x (torch.Tensor): [b, s, c]
 75 |         """
 76 | 
 77 |         if isinstance(x_or_tuple, tuple):
 78 |             # self.cfg.projector_type == "low_high_hybrid_split_mlp_gelu":
 79 |             high_x, low_x = x_or_tuple
 80 |             high_x = self.high_up_proj(high_x)
 81 |             low_x = self.low_up_proj(low_x)
 82 |             x = torch.concat([high_x, low_x], dim=-1)
 83 |         else:
 84 |             x = x_or_tuple
 85 | 
 86 |         return self.layers(x)
 87 | 
 88 | 
 89 | if __name__ == "__main__":
 90 |     cfg = AttrDict(
 91 |         input_dim=1024,
 92 |         n_embed=2048,
 93 |         depth=2,
 94 |         projector_type="low_high_hybrid_split_mlp_gelu",
 95 |     )
 96 |     inputs = (torch.rand(4, 576, 1024), torch.rand(4, 576, 1024))
 97 | 
 98 |     m = MlpProjector(cfg)
 99 |     out = m(inputs)
100 |     print(out.shape)
101 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/language_model/llava_mpt.py:
--------------------------------------------------------------------------------
  1 | #    Copyright 2023 Haotian Liu
  2 | #
  3 | #    Licensed under the Apache License, Version 2.0 (the "License");
  4 | #    you may not use this file except in compliance with the License.
  5 | #    You may obtain a copy of the License at
  6 | #
  7 | #        http://www.apache.org/licenses/LICENSE-2.0
  8 | #
  9 | #    Unless required by applicable law or agreed to in writing, software
 10 | #    distributed under the License is distributed on an "AS IS" BASIS,
 11 | #    WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
 12 | #    See the License for the specific language governing permissions and
 13 | #    limitations under the License.
 14 | 
 15 | 
 16 | from typing import Optional, Tuple
 17 | 
 18 | import torch
 19 | 
 20 | from transformers import AutoConfig, AutoModelForCausalLM, MptConfig, MptForCausalLM, MptModel, GenerationConfig
 21 | from llava.model.llava_arch import LlavaMetaModel, LlavaMetaForCausalLM
 22 | 
 23 | 
 24 | class LlavaMptConfig(MptConfig):
 25 |     model_type = "llava_mpt"
 26 | 
 27 | 
 28 | class LlavaMptModel(LlavaMetaModel, MptModel):
 29 |     config_class = LlavaMptConfig
 30 | 
 31 |     def __init__(self, config: MptConfig):
 32 |         config.hidden_size = config.d_model
 33 |         super(LlavaMptModel, self).__init__(config)
 34 | 
 35 |     def embed_tokens(self, x):
 36 |         return self.wte(x)
 37 | 
 38 | 
 39 | class LlavaMptForCausalLM(MptForCausalLM, LlavaMetaForCausalLM):
 40 |     config_class = LlavaMptConfig
 41 |     supports_gradient_checkpointing = True
 42 | 
 43 |     def __init__(self, config):
 44 |         super(MptForCausalLM, self).__init__(config)
 45 | 
 46 |         config.model_type = "llava_mpt"
 47 |         config.rope_scaling = None
 48 |         self.generation_config = GenerationConfig(
 49 |             temperature=0.0,
 50 |             max_new_tokens=1024,
 51 |             do_sample=False,
 52 |             top_p=None,
 53 |         )
 54 | 
 55 |         self.transformer = LlavaMptModel(config)
 56 |         self.lm_head = torch.nn.Linear(config.hidden_size, config.vocab_size, bias=False)
 57 | 
 58 |         # Initialize weights and apply final processing
 59 |         self.post_init()
 60 | 
 61 |     def get_model(self):
 62 |         return self.transformer
 63 | 
 64 |     def _set_gradient_checkpointing(self, module, value=False):
 65 |         if isinstance(module, LlavaMptModel):
 66 |             module.gradient_checkpointing = value
 67 | 
 68 |     def forward(
 69 |         self,
 70 |         input_ids: Optional[torch.LongTensor] = None,
 71 |         past_key_values: Optional[Tuple[Tuple[torch.Tensor, torch.Tensor], ...]] = None,
 72 |         attention_mask: Optional[torch.Tensor] = None,
 73 |         inputs_embeds: Optional[torch.Tensor] = None,
 74 |         labels: Optional[torch.Tensor] = None,
 75 |         use_cache: Optional[bool] = None,
 76 |         output_attentions: Optional[bool] = None,
 77 |         output_hidden_states: Optional[bool] = None,
 78 |         return_dict: Optional[bool] = None,
 79 |         cache_position=None,
 80 |         images=None,
 81 |     ):
 82 | 
 83 |         input_ids, attention_mask, past_key_values, inputs_embeds, labels = self.prepare_inputs_labels_for_multimodal(input_ids, attention_mask, past_key_values, labels, images)
 84 | 
 85 |         return super().forward(
 86 |             input_ids,
 87 |             past_key_values=past_key_values,
 88 |             attention_mask=attention_mask,
 89 |             inputs_embeds=inputs_embeds,
 90 |             labels=labels,
 91 |             use_cache=use_cache,
 92 |             output_attentions=output_attentions,
 93 |             output_hidden_states=output_hidden_states,
 94 |             return_dict=return_dict,
 95 |         )
 96 | 
 97 |     def prepare_inputs_for_generation(self, input_ids, past_key_values=None, inputs_embeds=None, **kwargs):
 98 |         images = kwargs.pop("images", None)
 99 |         _inputs = super().prepare_inputs_for_generation(input_ids, past_key_values=past_key_values, inputs_embeds=inputs_embeds, **kwargs)
100 |         _inputs["images"] = images
101 |         return _inputs
102 | 
103 | 
104 | AutoConfig.register("llava_mpt", LlavaMptConfig)
105 | AutoModelForCausalLM.register(LlavaMptConfig, LlavaMptForCausalLM)
106 | 


--------------------------------------------------------------------------------
/.gitignore:
--------------------------------------------------------------------------------
  1 | # Byte-compiled / optimized / DLL files
  2 | __pycache__/
  3 | *.py[cod]
  4 | *$py.class
  5 | 
  6 | # C extensions
  7 | *.so
  8 | 
  9 | # Distribution / packaging
 10 | .Python
 11 | build/
 12 | develop-eggs/
 13 | dist/
 14 | downloads/
 15 | eggs/
 16 | .eggs/
 17 | lib/
 18 | lib64/
 19 | parts/
 20 | sdist/
 21 | var/
 22 | wheels/
 23 | share/python-wheels/
 24 | *.egg-info/
 25 | .installed.cfg
 26 | *.egg
 27 | MANIFEST
 28 | 
 29 | # PyInstaller
 30 | #  Usually these files are written by a python script from a template
 31 | #  before PyInstaller builds the exe, so as to inject date/other infos into it.
 32 | *.manifest
 33 | *.spec
 34 | 
 35 | # Installer logs
 36 | pip-log.txt
 37 | pip-delete-this-directory.txt
 38 | 
 39 | # Unit test / coverage reports
 40 | htmlcov/
 41 | .tox/
 42 | .nox/
 43 | .coverage
 44 | .coverage.*
 45 | .cache
 46 | nosetests.xml
 47 | coverage.xml
 48 | *.cover
 49 | *.py,cover
 50 | .hypothesis/
 51 | .pytest_cache/
 52 | cover/
 53 | 
 54 | # Translations
 55 | *.mo
 56 | *.pot
 57 | 
 58 | # Django stuff:
 59 | *.log
 60 | local_settings.py
 61 | db.sqlite3
 62 | db.sqlite3-journal
 63 | 
 64 | # Flask stuff:
 65 | instance/
 66 | .webassets-cache
 67 | 
 68 | # Scrapy stuff:
 69 | .scrapy
 70 | 
 71 | # Sphinx documentation
 72 | docs/_build/
 73 | 
 74 | # PyBuilder
 75 | .pybuilder/
 76 | target/
 77 | 
 78 | # Jupyter Notebook
 79 | .ipynb_checkpoints
 80 | 
 81 | # IPython
 82 | profile_default/
 83 | ipython_config.py
 84 | 
 85 | # pyenv
 86 | #   For a library or package, you might want to ignore these files since the code is
 87 | #   intended to run in multiple environments; otherwise, check them in:
 88 | # .python-version
 89 | 
 90 | # pipenv
 91 | #   According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
 92 | #   However, in case of collaboration, if having platform-specific dependencies or dependencies
 93 | #   having no cross-platform support, pipenv may install dependencies that don't work, or not
 94 | #   install all needed dependencies.
 95 | #Pipfile.lock
 96 | 
 97 | # UV
 98 | #   Similar to Pipfile.lock, it is generally recommended to include uv.lock in version control.
 99 | #   This is especially recommended for binary packages to ensure reproducibility, and is more
100 | #   commonly ignored for libraries.
101 | #uv.lock
102 | 
103 | # poetry
104 | #   Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
105 | #   This is especially recommended for binary packages to ensure reproducibility, and is more
106 | #   commonly ignored for libraries.
107 | #   https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control
108 | #poetry.lock
109 | 
110 | # pdm
111 | #   Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
112 | #pdm.lock
113 | #   pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it
114 | #   in version control.
115 | #   https://pdm.fming.dev/latest/usage/project/#working-with-version-control
116 | .pdm.toml
117 | .pdm-python
118 | .pdm-build/
119 | 
120 | # PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm
121 | __pypackages__/
122 | 
123 | # Celery stuff
124 | celerybeat-schedule
125 | celerybeat.pid
126 | 
127 | # SageMath parsed files
128 | *.sage.py
129 | 
130 | # Environments
131 | .env
132 | .venv
133 | env/
134 | venv/
135 | ENV/
136 | env.bak/
137 | venv.bak/
138 | 
139 | # Spyder project settings
140 | .spyderproject
141 | .spyproject
142 | 
143 | # Rope project settings
144 | .ropeproject
145 | 
146 | # mkdocs documentation
147 | /site
148 | 
149 | # mypy
150 | .mypy_cache/
151 | .dmypy.json
152 | dmypy.json
153 | 
154 | # Pyre type checker
155 | .pyre/
156 | 
157 | # pytype static type analyzer
158 | .pytype/
159 | 
160 | # Cython debug symbols
161 | cython_debug/
162 | 
163 | # PyCharm
164 | #  JetBrains specific template is maintained in a separate JetBrains.gitignore that can
165 | #  be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore
166 | #  and can be added to the global gitignore or merged into this file.  For a more nuclear
167 | #  option (not recommended) you can uncomment the following to ignore the entire idea folder.
168 | #.idea/
169 | 
170 | # PyPI configuration file
171 | .pypirc
172 | 
173 | # Temp folders
174 | wandb/
175 | checkpoints/
176 | .vscode/
177 | outputs/
178 | 
179 | *.pth
180 | *.pt
181 | src/r1-v/src/utils/experts/expert_weights
182 | debug.sh


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/train/llama_flash_attn_monkey_patch.py:
--------------------------------------------------------------------------------
 1 | from typing import Optional, Tuple
 2 | import warnings
 3 | 
 4 | import torch
 5 | 
 6 | import transformers
 7 | from transformers.models.llama.modeling_llama import apply_rotary_pos_emb, repeat_kv
 8 | 
 9 | try:
10 |     from flash_attn.flash_attn_interface import flash_attn_unpadded_qkvpacked_func
11 | except ImportError:
12 |     from flash_attn.flash_attn_interface import flash_attn_varlen_qkvpacked_func as flash_attn_unpadded_qkvpacked_func
13 | from flash_attn.bert_padding import unpad_input, pad_input
14 | 
15 | 
16 | def forward(
17 |     self,
18 |     hidden_states: torch.Tensor,
19 |     attention_mask: Optional[torch.Tensor] = None,
20 |     position_ids: Optional[torch.Tensor] = None,
21 |     past_key_value: Optional[Tuple[torch.Tensor]] = None,
22 |     output_attentions: bool = False,
23 |     use_cache: bool = False,
24 |     padding_mask: Optional[torch.Tensor] = None,
25 | ) -> Tuple[torch.Tensor, Optional[torch.Tensor], Optional[Tuple[torch.Tensor]]]:
26 |     if output_attentions:
27 |         warnings.warn("Output attentions is not supported for patched `LlamaAttention`, returning `None` instead.")
28 | 
29 |     bsz, q_len, _ = hidden_states.size()
30 | 
31 |     query_states = self.q_proj(hidden_states).view(bsz, q_len, self.num_heads, self.head_dim).transpose(1, 2)
32 |     key_states = self.k_proj(hidden_states).view(bsz, q_len, self.num_key_value_heads, self.head_dim).transpose(1, 2)
33 |     value_states = self.v_proj(hidden_states).view(bsz, q_len, self.num_key_value_heads, self.head_dim).transpose(1, 2)  # shape: (b, num_heads, s, head_dim)
34 | 
35 |     kv_seq_len = key_states.shape[-2]
36 |     if past_key_value is not None:
37 |         kv_seq_len += past_key_value[0].shape[-2]
38 | 
39 |     cos, sin = self.rotary_emb(value_states, seq_len=kv_seq_len)
40 |     query_states, key_states = apply_rotary_pos_emb(query_states, key_states, cos, sin, position_ids)
41 | 
42 |     if past_key_value is not None:
43 |         # reuse k, v
44 |         key_states = torch.cat([past_key_value[0], key_states], dim=2)
45 |         value_states = torch.cat([past_key_value[1], value_states], dim=2)
46 | 
47 |     past_key_value = (key_states, value_states) if use_cache else None
48 | 
49 |     # repeat k/v heads if n_kv_heads < n_heads
50 |     key_states = repeat_kv(key_states, self.num_key_value_groups)
51 |     value_states = repeat_kv(value_states, self.num_key_value_groups)
52 | 
53 |     # Transform the data into the format required by flash attention
54 |     qkv = torch.stack([query_states, key_states, value_states], dim=2)
55 |     qkv = qkv.transpose(1, 3)  # shape: [b, s, 3, num_heads, head_dim]
56 |     key_padding_mask = attention_mask
57 | 
58 |     if key_padding_mask is None:
59 |         qkv = qkv.reshape(-1, 3, self.num_heads, self.head_dim)
60 |         cu_q_lens = torch.arange(0, (bsz + 1) * q_len, step=q_len, dtype=torch.int32, device=qkv.device)
61 |         max_s = q_len
62 |         output = flash_attn_unpadded_qkvpacked_func(qkv, cu_q_lens, max_s, 0.0, softmax_scale=None, causal=True)
63 |         output = output.view(bsz, q_len, -1)
64 |     else:
65 |         qkv = qkv.reshape(bsz, q_len, -1)
66 |         qkv, indices, cu_q_lens, max_s = unpad_input(qkv, key_padding_mask)
67 |         qkv = qkv.view(-1, 3, self.num_heads, self.head_dim)
68 |         output_unpad = flash_attn_unpadded_qkvpacked_func(qkv, cu_q_lens, max_s, 0.0, softmax_scale=None, causal=True)
69 |         output_unpad = output_unpad.reshape(-1, self.num_heads * self.head_dim)
70 |         output = pad_input(output_unpad, indices, bsz, q_len)
71 | 
72 |     return self.o_proj(output), None, past_key_value
73 | 
74 | 
75 | # Disable the transformation of the attention mask in LlamaModel as the flash attention
76 | # requires the attention mask to be the same as the key_padding_mask
77 | def _prepare_decoder_attention_mask(self, attention_mask, input_shape, inputs_embeds, past_key_values_length):
78 |     # [bsz, seq_len]
79 |     return attention_mask
80 | 
81 | 
82 | def replace_llama_attn_with_flash_attn():
83 |     cuda_major, cuda_minor = torch.cuda.get_device_capability()
84 |     if cuda_major < 8:
85 |         warnings.warn("Flash attention is only supported on A100 or H100 GPU during training due to head dim > 64 backward." "ref: https://github.com/HazyResearch/flash-attention/issues/190#issuecomment-1523359593")
86 |     transformers.models.llama.modeling_llama.LlamaModel._prepare_decoder_attention_mask = _prepare_decoder_attention_mask
87 |     transformers.models.llama.modeling_llama.LlamaAttention.forward = forward
88 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/groundingdino/util/box_ops.py:
--------------------------------------------------------------------------------
  1 | # Copyright (c) Facebook, Inc. and its affiliates. All Rights Reserved
  2 | """
  3 | Utilities for bounding box manipulation and GIoU.
  4 | """
  5 | import torch
  6 | from torchvision.ops.boxes import box_area
  7 | 
  8 | 
  9 | def box_cxcywh_to_xyxy(x):
 10 |     x_c, y_c, w, h = x.unbind(-1)
 11 |     b = [(x_c - 0.5 * w), (y_c - 0.5 * h), (x_c + 0.5 * w), (y_c + 0.5 * h)]
 12 |     return torch.stack(b, dim=-1)
 13 | 
 14 | 
 15 | def box_xyxy_to_cxcywh(x):
 16 |     x0, y0, x1, y1 = x.unbind(-1)
 17 |     b = [(x0 + x1) / 2, (y0 + y1) / 2, (x1 - x0), (y1 - y0)]
 18 |     return torch.stack(b, dim=-1)
 19 | 
 20 | 
 21 | # modified from torchvision to also return the union
 22 | def box_iou(boxes1, boxes2):
 23 |     area1 = box_area(boxes1)
 24 |     area2 = box_area(boxes2)
 25 | 
 26 |     # import ipdb; ipdb.set_trace()
 27 |     lt = torch.max(boxes1[:, None, :2], boxes2[:, :2])  # [N,M,2]
 28 |     rb = torch.min(boxes1[:, None, 2:], boxes2[:, 2:])  # [N,M,2]
 29 | 
 30 |     wh = (rb - lt).clamp(min=0)  # [N,M,2]
 31 |     inter = wh[:, :, 0] * wh[:, :, 1]  # [N,M]
 32 | 
 33 |     union = area1[:, None] + area2 - inter
 34 | 
 35 |     iou = inter / (union + 1e-6)
 36 |     return iou, union
 37 | 
 38 | 
 39 | def generalized_box_iou(boxes1, boxes2):
 40 |     """
 41 |     Generalized IoU from https://giou.stanford.edu/
 42 | 
 43 |     The boxes should be in [x0, y0, x1, y1] format
 44 | 
 45 |     Returns a [N, M] pairwise matrix, where N = len(boxes1)
 46 |     and M = len(boxes2)
 47 |     """
 48 |     # degenerate boxes gives inf / nan results
 49 |     # so do an early check
 50 |     assert (boxes1[:, 2:] >= boxes1[:, :2]).all()
 51 |     assert (boxes2[:, 2:] >= boxes2[:, :2]).all()
 52 |     # except:
 53 |     #     import ipdb; ipdb.set_trace()
 54 |     iou, union = box_iou(boxes1, boxes2)
 55 | 
 56 |     lt = torch.min(boxes1[:, None, :2], boxes2[:, :2])
 57 |     rb = torch.max(boxes1[:, None, 2:], boxes2[:, 2:])
 58 | 
 59 |     wh = (rb - lt).clamp(min=0)  # [N,M,2]
 60 |     area = wh[:, :, 0] * wh[:, :, 1]
 61 | 
 62 |     return iou - (area - union) / (area + 1e-6)
 63 | 
 64 | 
 65 | # modified from torchvision to also return the union
 66 | def box_iou_pairwise(boxes1, boxes2):
 67 |     area1 = box_area(boxes1)
 68 |     area2 = box_area(boxes2)
 69 | 
 70 |     lt = torch.max(boxes1[:, :2], boxes2[:, :2])  # [N,2]
 71 |     rb = torch.min(boxes1[:, 2:], boxes2[:, 2:])  # [N,2]
 72 | 
 73 |     wh = (rb - lt).clamp(min=0)  # [N,2]
 74 |     inter = wh[:, 0] * wh[:, 1]  # [N]
 75 | 
 76 |     union = area1 + area2 - inter
 77 | 
 78 |     iou = inter / union
 79 |     return iou, union
 80 | 
 81 | 
 82 | def generalized_box_iou_pairwise(boxes1, boxes2):
 83 |     """
 84 |     Generalized IoU from https://giou.stanford.edu/
 85 | 
 86 |     Input:
 87 |         - boxes1, boxes2: N,4
 88 |     Output:
 89 |         - giou: N, 4
 90 |     """
 91 |     # degenerate boxes gives inf / nan results
 92 |     # so do an early check
 93 |     assert (boxes1[:, 2:] >= boxes1[:, :2]).all()
 94 |     assert (boxes2[:, 2:] >= boxes2[:, :2]).all()
 95 |     assert boxes1.shape == boxes2.shape
 96 |     iou, union = box_iou_pairwise(boxes1, boxes2)  # N, 4
 97 | 
 98 |     lt = torch.min(boxes1[:, :2], boxes2[:, :2])
 99 |     rb = torch.max(boxes1[:, 2:], boxes2[:, 2:])
100 | 
101 |     wh = (rb - lt).clamp(min=0)  # [N,2]
102 |     area = wh[:, 0] * wh[:, 1]
103 | 
104 |     return iou - (area - union) / area
105 | 
106 | 
107 | def masks_to_boxes(masks):
108 |     """Compute the bounding boxes around the provided masks
109 | 
110 |     The masks should be in format [N, H, W] where N is the number of masks, (H, W) are the spatial dimensions.
111 | 
112 |     Returns a [N, 4] tensors, with the boxes in xyxy format
113 |     """
114 |     if masks.numel() == 0:
115 |         return torch.zeros((0, 4), device=masks.device)
116 | 
117 |     h, w = masks.shape[-2:]
118 | 
119 |     y = torch.arange(0, h, dtype=torch.float)
120 |     x = torch.arange(0, w, dtype=torch.float)
121 |     y, x = torch.meshgrid(y, x)
122 | 
123 |     x_mask = masks * x.unsqueeze(0)
124 |     x_max = x_mask.flatten(1).max(-1)[0]
125 |     x_min = x_mask.masked_fill(~(masks.bool()), 1e8).flatten(1).min(-1)[0]
126 | 
127 |     y_mask = masks * y.unsqueeze(0)
128 |     y_max = y_mask.flatten(1).max(-1)[0]
129 |     y_min = y_mask.masked_fill(~(masks.bool()), 1e8).flatten(1).min(-1)[0]
130 | 
131 |     return torch.stack([x_min, y_min, x_max, y_max], 1)
132 | 
133 | 
134 | if __name__ == "__main__":
135 |     x = torch.rand(5, 4)
136 |     y = torch.rand(3, 4)
137 |     iou, union = box_iou(x, y)
138 |     import ipdb
139 | 
140 |     ipdb.set_trace()
141 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/GroundingDINO/groundingdino/models/GroundingDINO/transformer_vanilla.py:
--------------------------------------------------------------------------------
  1 | # ------------------------------------------------------------------------
  2 | # Grounding DINO
  3 | # url: https://github.com/IDEA-Research/GroundingDINO
  4 | # Copyright (c) 2023 IDEA. All Rights Reserved.
  5 | # Licensed under the Apache License, Version 2.0 [see LICENSE for details]
  6 | # ------------------------------------------------------------------------
  7 | # Copyright (c) Aishwarya Kamath & Nicolas Carion. Licensed under the Apache License 2.0. All Rights Reserved
  8 | # Copyright (c) Facebook, Inc. and its affiliates. All Rights Reserved
  9 | """
 10 | DETR Transformer class.
 11 | 
 12 | Copy-paste from torch.nn.Transformer with modifications:
 13 |     * positional encodings are passed in MHattention
 14 |     * extra LN at the end of encoder is removed
 15 |     * decoder returns a stack of activations from all decoding layers
 16 | """
 17 | from typing import Optional
 18 | 
 19 | import torch
 20 | import torch.nn.functional as F
 21 | from torch import Tensor, nn
 22 | 
 23 | from .utils import (
 24 |     MLP,
 25 |     _get_activation_fn,
 26 |     _get_clones,
 27 |     gen_encoder_output_proposals,
 28 |     gen_sineembed_for_position,
 29 |     sigmoid_focal_loss,
 30 | )
 31 | 
 32 | 
 33 | class TextTransformer(nn.Module):
 34 |     def __init__(self, num_layers, d_model=256, nheads=8, dim_feedforward=2048, dropout=0.1):
 35 |         super().__init__()
 36 |         self.num_layers = num_layers
 37 |         self.d_model = d_model
 38 |         self.nheads = nheads
 39 |         self.dim_feedforward = dim_feedforward
 40 |         self.norm = None
 41 | 
 42 |         single_encoder_layer = TransformerEncoderLayer(
 43 |             d_model=d_model, nhead=nheads, dim_feedforward=dim_feedforward, dropout=dropout
 44 |         )
 45 |         self.layers = _get_clones(single_encoder_layer, num_layers)
 46 | 
 47 |     def forward(self, memory_text: torch.Tensor, text_attention_mask: torch.Tensor):
 48 |         """
 49 | 
 50 |         Args:
 51 |             text_attention_mask: bs, num_token
 52 |             memory_text: bs, num_token, d_model
 53 | 
 54 |         Raises:
 55 |             RuntimeError: _description_
 56 | 
 57 |         Returns:
 58 |             output: bs, num_token, d_model
 59 |         """
 60 | 
 61 |         output = memory_text.transpose(0, 1)
 62 | 
 63 |         for layer in self.layers:
 64 |             output = layer(output, src_key_padding_mask=text_attention_mask)
 65 | 
 66 |         if self.norm is not None:
 67 |             output = self.norm(output)
 68 | 
 69 |         return output.transpose(0, 1)
 70 | 
 71 | 
 72 | class TransformerEncoderLayer(nn.Module):
 73 |     def __init__(
 74 |         self,
 75 |         d_model,
 76 |         nhead,
 77 |         dim_feedforward=2048,
 78 |         dropout=0.1,
 79 |         activation="relu",
 80 |         normalize_before=False,
 81 |     ):
 82 |         super().__init__()
 83 |         self.self_attn = nn.MultiheadAttention(d_model, nhead, dropout=dropout)
 84 |         # Implementation of Feedforward model
 85 |         self.linear1 = nn.Linear(d_model, dim_feedforward)
 86 |         self.dropout = nn.Dropout(dropout)
 87 |         self.linear2 = nn.Linear(dim_feedforward, d_model)
 88 | 
 89 |         self.norm1 = nn.LayerNorm(d_model)
 90 |         self.norm2 = nn.LayerNorm(d_model)
 91 |         self.dropout1 = nn.Dropout(dropout)
 92 |         self.dropout2 = nn.Dropout(dropout)
 93 | 
 94 |         self.activation = _get_activation_fn(activation)
 95 |         self.normalize_before = normalize_before
 96 |         self.nhead = nhead
 97 | 
 98 |     def with_pos_embed(self, tensor, pos: Optional[Tensor]):
 99 |         return tensor if pos is None else tensor + pos
100 | 
101 |     def forward(
102 |         self,
103 |         src,
104 |         src_mask: Optional[Tensor] = None,
105 |         src_key_padding_mask: Optional[Tensor] = None,
106 |         pos: Optional[Tensor] = None,
107 |     ):
108 |         # repeat attn mask
109 |         if src_mask.dim() == 3 and src_mask.shape[0] == src.shape[1]:
110 |             # bs, num_q, num_k
111 |             src_mask = src_mask.repeat(self.nhead, 1, 1)
112 | 
113 |         q = k = self.with_pos_embed(src, pos)
114 | 
115 |         src2 = self.self_attn(q, k, value=src, attn_mask=src_mask)[0]
116 | 
117 |         # src2 = self.self_attn(q, k, value=src, attn_mask=src_mask, key_padding_mask=src_key_padding_mask)[0]
118 |         src = src + self.dropout1(src2)
119 |         src = self.norm1(src)
120 |         src2 = self.linear2(self.dropout(self.activation(self.linear1(src))))
121 |         src = src + self.dropout2(src2)
122 |         src = self.norm2(src)
123 |         return src
124 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/reward_hps.py:
--------------------------------------------------------------------------------
  1 | import os
  2 | import os
  3 | import torch
  4 | import requests
  5 | from PIL import Image
  6 | from hpsv2.src.open_clip import create_model_and_transforms, get_tokenizer
  7 | 
  8 | class HPSv2:
  9 |     def __init__(self, args):
 10 |         self.ckpt_path = args.hps_ckpt_path
 11 | 
 12 |     @property
 13 |     def __name__(self):
 14 |         return 'HPSv2'
 15 |     
 16 |     def load_to_device(self, load_device):
 17 |         self.model, self.preprocess_train, self.preprocess_val = create_model_and_transforms(
 18 |                     'ViT-H-14',
 19 |                     pretrained='laion2b_s32b_b79k',
 20 |                     precision='amp',
 21 |                     device=load_device,
 22 |                     jit=False,
 23 |                     force_quick_gelu=False,
 24 |                     force_custom_text=False,
 25 |                     force_patch_dropout=False,
 26 |                     force_image_size=None,
 27 |                     pretrained_image=False,
 28 |                     image_mean=None,
 29 |                     image_std=None,
 30 |                     light_augmentation=True,
 31 |                     aug_cfg={},
 32 |                     output_dict=True,
 33 |                     with_score_predictor=False,
 34 |                     with_region_predictor=False
 35 |                 )
 36 |         # workaround for the zero3
 37 |         checkpoint = torch.load(self.ckpt_path, map_location='cpu')
 38 |         self.model.load_state_dict(checkpoint['state_dict'])
 39 |         for param in self.model.parameters():
 40 |             param.requires_grad = False
 41 |         
 42 |         self.tokenizer = get_tokenizer('ViT-H-14')
 43 |         self.model = self.model.to(load_device)
 44 |         self.model.eval()
 45 |     
 46 |     def __call__(self, prompts, images, **kwargs):
 47 |         # image_list is a list of PIL image
 48 |         device = list(self.model.parameters())[0].device
 49 |         result = []
 50 |         for i, (prompt, image) in enumerate(zip(prompts, images)):
 51 | 
 52 |             with torch.no_grad():
 53 |                 # Process the image
 54 |                 image = self.preprocess_val(image).unsqueeze(0).to(device=device, non_blocking=True)
 55 |                 # Process the prompt
 56 |                 text = self.tokenizer([prompt]).to(device=device, non_blocking=True)
 57 |                 # Calculate the HPS
 58 |                 with torch.amp.autocast(device_type='cuda'):
 59 |                     outputs = self.model(image, text)
 60 |                     image_features, text_features = outputs["image_features"], outputs["text_features"]
 61 |                     logits_per_image = image_features @ text_features.T
 62 | 
 63 |                     hps_score = torch.diagonal(logits_per_image).cpu().numpy()
 64 |             result.append(hps_score[0])
 65 |         return result
 66 | 
 67 | class HPSv2Compare(HPSv2):
 68 |     def __init__(self):
 69 |         super().__init__()
 70 |     
 71 |     @property
 72 |     def __name__(self):
 73 |         return 'HPSv2Compare'
 74 | 
 75 |     def __call__(self, prompts, images, image_path, **kwargs): 
 76 |         
 77 |         image_before_list = [Image.open(i) for i in image_path]
 78 |         # image_list is a list of PIL image
 79 |         device = list(self.model.parameters())[0].device
 80 |         result = []
 81 |         for prompt, image, image_before in zip(prompts, images, image_before_list):
 82 |             with torch.no_grad():
 83 |                 # Process the image
 84 |                 image = self.preprocess_val(image).unsqueeze(0).to(device=device, non_blocking=True)
 85 |                 image_before = self.preprocess_val(image_before).unsqueeze(0).to(device=device, non_blocking=True)
 86 |                 # Process the prompt
 87 |                 text = self.tokenizer([prompt]).to(device=device, non_blocking=True)
 88 |                 # Calculate the HPS
 89 |                 with torch.amp.autocast(device_type='cuda'):
 90 |                     outputs = self.model(image, text)
 91 |                     image_features, text_features = outputs["image_features"], outputs["text_features"]
 92 |                     logits_per_image = image_features @ text_features.T
 93 |                     hps_score = torch.diagonal(logits_per_image).cpu().numpy()
 94 | 
 95 |                     outputs_before = self.model(image_before, text)
 96 |                     image_features_before, text_features_before = outputs_before["image_features"], outputs_before["text_features"]
 97 |                     logits_per_image_before = image_features_before @ text_features_before.T
 98 |                     hps_score_before = torch.diagonal(logits_per_image_before).cpu().numpy()
 99 |             result.append(hps_score[0] - hps_score_before[0])
100 |         return result


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/serve/cli.py:
--------------------------------------------------------------------------------
  1 | import argparse
  2 | import torch
  3 | 
  4 | from llava.constants import IMAGE_TOKEN_INDEX, DEFAULT_IMAGE_TOKEN, DEFAULT_IM_START_TOKEN, DEFAULT_IM_END_TOKEN
  5 | from llava.conversation import conv_templates, SeparatorStyle
  6 | from llava.model.builder import load_pretrained_model
  7 | from llava.utils import disable_torch_init
  8 | from llava.mm_utils import tokenizer_image_token, get_model_name_from_path, KeywordsStoppingCriteria
  9 | 
 10 | from PIL import Image
 11 | 
 12 | import requests
 13 | from PIL import Image
 14 | from io import BytesIO
 15 | from transformers import TextStreamer
 16 | 
 17 | 
 18 | def load_image(image_file):
 19 |     if image_file.startswith("http") or image_file.startswith("https"):
 20 |         response = requests.get(image_file)
 21 |         image = Image.open(BytesIO(response.content)).convert("RGB")
 22 |     else:
 23 |         image = Image.open(image_file).convert("RGB")
 24 |     return image
 25 | 
 26 | 
 27 | def main(args):
 28 |     # Model
 29 |     disable_torch_init()
 30 | 
 31 |     model_name = get_model_name_from_path(args.model_path)
 32 |     tokenizer, model, image_processor, context_len = load_pretrained_model(args.model_path, args.model_base, model_name, args.load_8bit, args.load_4bit)
 33 | 
 34 |     if "llama-2" in model_name.lower():
 35 |         conv_mode = "llava_llama_2"
 36 |     elif "v1" in model_name.lower():
 37 |         conv_mode = "llava_v1"
 38 |     elif "mpt" in model_name.lower():
 39 |         conv_mode = "mpt"
 40 |     else:
 41 |         conv_mode = "llava_v0"
 42 | 
 43 |     if args.conv_mode is not None and conv_mode != args.conv_mode:
 44 |         print("[WARNING] the auto inferred conversation mode is {}, while `--conv-mode` is {}, using {}".format(conv_mode, args.conv_mode, args.conv_mode))
 45 |     else:
 46 |         args.conv_mode = conv_mode
 47 | 
 48 |     conv = conv_templates[args.conv_mode].copy()
 49 |     if "mpt" in model_name.lower():
 50 |         roles = ("user", "assistant")
 51 |     else:
 52 |         roles = conv.roles
 53 | 
 54 |     image = load_image(args.image_file)
 55 |     image_tensor = image_processor.preprocess(image, return_tensors="pt")["pixel_values"].half().cuda()
 56 | 
 57 |     while True:
 58 |         try:
 59 |             inp = input(f"{roles[0]}: ")
 60 |         except EOFError:
 61 |             inp = ""
 62 |         if not inp:
 63 |             print("exit...")
 64 |             break
 65 | 
 66 |         print(f"{roles[1]}: ", end="")
 67 | 
 68 |         if image is not None:
 69 |             # first message
 70 |             if model.config.mm_use_im_start_end:
 71 |                 inp = DEFAULT_IM_START_TOKEN + DEFAULT_IMAGE_TOKEN + DEFAULT_IM_END_TOKEN + "\n" + inp
 72 |             else:
 73 |                 inp = DEFAULT_IMAGE_TOKEN + "\n" + inp
 74 |             conv.append_message(conv.roles[0], inp)
 75 |             image = None
 76 |         else:
 77 |             # later messages
 78 |             conv.append_message(conv.roles[0], inp)
 79 |         conv.append_message(conv.roles[1], None)
 80 |         prompt = conv.get_prompt()
 81 | 
 82 |         input_ids = tokenizer_image_token(prompt, tokenizer, IMAGE_TOKEN_INDEX, return_tensors="pt").unsqueeze(0).cuda()
 83 |         stop_str = conv.sep if conv.sep_style != SeparatorStyle.TWO else conv.sep2
 84 |         keywords = [stop_str]
 85 |         stopping_criteria = KeywordsStoppingCriteria(keywords, tokenizer, input_ids)
 86 |         streamer = TextStreamer(tokenizer, skip_prompt=True, skip_special_tokens=True)
 87 | 
 88 |         with torch.inference_mode():
 89 |             output_ids = model.generate(input_ids, images=image_tensor, do_sample=True, temperature=0.2, max_new_tokens=1024, streamer=streamer, use_cache=True, stopping_criteria=[stopping_criteria])
 90 | 
 91 |         outputs = tokenizer.decode(output_ids[0, input_ids.shape[1] :]).strip()
 92 |         conv.messages[-1][-1] = outputs
 93 | 
 94 |         if args.debug:
 95 |             print("\n", {"prompt": prompt, "outputs": outputs}, "\n")
 96 | 
 97 | 
 98 | if __name__ == "__main__":
 99 |     parser = argparse.ArgumentParser()
100 |     parser.add_argument("--model-path", type=str, default="facebook/opt-350m")
101 |     parser.add_argument("--model-base", type=str, default=None)
102 |     parser.add_argument("--image-file", type=str, required=True)
103 |     parser.add_argument("--num-gpus", type=int, default=1)
104 |     parser.add_argument("--conv-mode", type=str, default=None)
105 |     parser.add_argument("--temperature", type=float, default=0.2)
106 |     parser.add_argument("--max-new-tokens", type=int, default=512)
107 |     parser.add_argument("--load-8bit", action="store_true")
108 |     parser.add_argument("--load-4bit", action="store_true")
109 |     parser.add_argument("--debug", action="store_true")
110 |     args = parser.parse_args()
111 |     main(args)
112 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/hf_vision.py:
--------------------------------------------------------------------------------
  1 | import torch
  2 | import torch.nn as nn
  3 | 
  4 | from transformers import AutoModel, AutoImageProcessor, AutoConfig, CLIPImageProcessor
  5 | from llava.utils import rank0_print
  6 | 
  7 | 
  8 | class HFVisionTower(nn.Module):
  9 |     def __init__(self, vision_tower, args, delay_load=False):
 10 |         super().__init__()
 11 | 
 12 |         self.is_loaded = False
 13 | 
 14 |         self.vision_tower_name = vision_tower.replace("hf:", "", 1)
 15 |         self.select_layer = args.mm_vision_select_layer
 16 |         self.select_feature = getattr(args, "mm_vision_select_feature", "patch")
 17 | 
 18 |         if not delay_load:
 19 |             self.load_model()
 20 |         else:
 21 |             self.cfg_only = AutoConfig.from_pretrained(self.vision_tower_name)
 22 | 
 23 |     def load_model(self):
 24 |         try:
 25 |             self.image_processor = AutoImageProcessor.from_pretrained(self.vision_tower_name)
 26 |         except Exception as e:
 27 |             if "448" in self.vision_tower_name:
 28 |                 image_size = 448
 29 |                 # use image processor with conig
 30 |                 self.image_processor = CLIPImageProcessor(size={"shortest_edge": image_size}, do_center_crop=True, crop_size=image_size)
 31 |             else:
 32 |                 self.image_processor = CLIPImageProcessor.from_pretrained("openai/clip-vit-large-patch14")
 33 |         rank0_print(f"Loaded image processor: {self.image_processor}")
 34 |         self.vision_tower = AutoModel.from_pretrained(self.vision_tower_name, torch_dtype=torch.bfloat16, trust_remote_code=True).to("cuda")
 35 |         self.device = self.vision_tower.device
 36 |         self.dtype = self.vision_tower.dtype
 37 |         self.config = self.vision_tower.config
 38 | 
 39 |         if hasattr(self.vision_tower, "vision_model"):
 40 |             self.vision_tower = self.vision_tower.vision_model
 41 |         self.vision_tower.requires_grad_(False)
 42 |         # self.vision_tower.eval()
 43 |         self.is_loaded = True
 44 | 
 45 |     def feature_select(self, image_forward_outs):
 46 |         select_feature_type = self.select_feature
 47 | 
 48 |         if self.select_feature in ["slicefour_patch", "slicefour_cls_patch"]:
 49 |             select_every_k_layer = len(image_forward_outs.hidden_states) // 4
 50 |             image_features = torch.cat([image_forward_outs.hidden_states[i] for i in range(select_every_k_layer + self.select_layer, len(image_forward_outs.hidden_states), select_every_k_layer)], dim=-1)
 51 |             select_feature_type = select_feature_type.replace("slicefour_", "")
 52 |         else:
 53 |             image_features = image_forward_outs.hidden_states[self.select_layer]
 54 | 
 55 |         if select_feature_type == "patch":
 56 |             image_features = image_features[:, 1:]
 57 |         elif select_feature_type == "cls_patch":
 58 |             image_features = image_features
 59 |         else:
 60 |             raise ValueError(f"Unexpected select feature: {select_feature_type}")
 61 |         return image_features
 62 | 
 63 |     def forward(self, images):
 64 |         if type(images) is list:
 65 |             image_features = []
 66 |             for image in images:
 67 |                 image_forward_out = self.vision_tower(image.to(device=self.device, dtype=self.dtype).unsqueeze(0), output_hidden_states=True)
 68 |                 image_feature = self.feature_select(image_forward_out).to(image.dtype)
 69 |                 image_features.append(image_feature)
 70 |         else:
 71 |             image_forward_outs = self.vision_tower(images.to(device=self.device, dtype=self.dtype), output_hidden_states=True)
 72 |             image_features = self.feature_select(image_forward_outs).to(images.dtype)
 73 | 
 74 |         return image_features
 75 | 
 76 |     @property
 77 |     def dummy_feature(self):
 78 |         return torch.zeros(1, self.hidden_size, device=self.device, dtype=self.dtype)
 79 | 
 80 |     # @property
 81 |     # def dtype(self):
 82 |     #     return self.vision_tower.dtype
 83 | 
 84 |     # @property
 85 |     # def device(self):
 86 |     #     return self.vision_tower.device
 87 | 
 88 |     @property
 89 |     def hidden_size(self):
 90 |         try:
 91 |             _hidden_size = self.config.hidden_size
 92 |         except:
 93 |             _hidden_size = self.config.vision_config.hidden_size
 94 |         if "slicefour" in self.select_feature:
 95 |             _hidden_size *= 4
 96 |         return _hidden_size
 97 | 
 98 |     @property
 99 |     def num_patches(self):
100 |         _num_patches = (self.config.image_size // self.config.patch_size) ** 2
101 |         if "cls_patch" in self.select_feature:
102 |             _num_patches += 1
103 |         return _num_patches
104 | 
105 |     @property
106 |     def num_patches_per_side(self):
107 |         return self.config.image_size // self.config.patch_size
108 | 
109 |     @property
110 |     def image_size(self):
111 |         return self.config.image_size
112 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/reward_orm.py:
--------------------------------------------------------------------------------
  1 | import os
  2 | 
  3 | import torch
  4 | import requests
  5 | from PIL import Image
  6 | import numpy as np
  7 | import copy
  8 | import argparse
  9 | import sys
 10 | 
 11 | from llava.model.builder import load_pretrained_model
 12 | from llava.mm_utils import process_images, tokenizer_image_token
 13 | from llava.constants import IMAGE_TOKEN_INDEX, DEFAULT_IMAGE_TOKEN
 14 | from llava.conversation import conv_templates
 15 | 
 16 | 
 17 | class ORM:
 18 |     def __init__(self, args):
 19 | 
 20 |         ckpt_path = args.orm_ckpt_path
 21 | 
 22 |         # Load model
 23 |         llava_model_args = {"multimodal": True}
 24 | 
 25 |         overwrite_config = {"image_aspect_ratio": "pad", "torch_dtype": 'bfloat16'}
 26 |         llava_model_args["overwrite_config"] = overwrite_config
 27 |         
 28 |         pretrained = ckpt_path
 29 |         print(f"pretrained path:{pretrained}")
 30 |         model_name = "llava_qwen"
 31 |         device_map = "cpu"
 32 |         self.tokenizer, self.model, self.image_processor, _ = load_pretrained_model(
 33 |             pretrained, None, model_name, device_map=device_map, torch_dtype='bfloat16', **llava_model_args
 34 |         )
 35 |         self.config = self.model.config
 36 | 
 37 |         self.model.eval()
 38 | 
 39 |         self.yes_token_id = self.tokenizer.convert_tokens_to_ids("yes")
 40 |         self.no_token_id = self.tokenizer.convert_tokens_to_ids("no")
 41 | 
 42 |     @property
 43 |     def __name__(self):
 44 |         return 'ORM'
 45 |     
 46 |     def load_to_device(self, load_device):
 47 |         self.device = load_device
 48 | 
 49 |         # freeze all parameters
 50 |         # for n, p in self.model.named_parameters():
 51 |         #     p.requires_grad = False
 52 |         # self.model.eval()
 53 | 
 54 |     def __call__(self, prompts, images, **kwargs):
 55 | 
 56 |         # Load the image
 57 |         results = []
 58 |         for prompt, image in zip(prompts, images):
 59 | 
 60 |             # Process the image
 61 |             image_tensor = process_images([image], self.image_processor, self.config)[0]
 62 |             image_tensor = image_tensor.to(dtype=torch.bfloat16, device=self.device)
 63 | 
 64 |             question = (f"{DEFAULT_IMAGE_TOKEN} This image is generated by a prompt: {prompt}. Does this image accurately represent the prompt? Please answer yes or no without explanation.")
 65 | 
 66 |             # Prepare conversation
 67 |             conv_template = "qwen_1_5"
 68 |             conv = copy.deepcopy(conv_templates[conv_template])
 69 |             conv.append_message(conv.roles[0], question)
 70 |             conv.append_message(conv.roles[1], None)
 71 |             prompt_question = conv.get_prompt()
 72 | 
 73 |             # Input question and image to the model
 74 |             input_ids = tokenizer_image_token(prompt_question, self.tokenizer, IMAGE_TOKEN_INDEX, return_tensors="pt").unsqueeze(0).to(self.device)
 75 |             image_size = image.size
 76 | 
 77 |             succeed = False
 78 |             max_retries = 1
 79 |             retry_count = 0
 80 | 
 81 | 
 82 |             while not succeed and retry_count < max_retries:
 83 |                 retry_count += 1
 84 |                 # Generate answer
 85 |                 with torch.amp.autocast(device_type='cuda', dtype=torch.bfloat16):
 86 |                     with torch.no_grad():
 87 |                         cont = self.model.generate(
 88 |                             input_ids,
 89 |                             images=[image_tensor],
 90 |                             image_sizes=[image_size],
 91 |                             do_sample=True,
 92 |                             temperature=0.3,
 93 |                             max_new_tokens=100,
 94 |                             return_dict_in_generate=True,
 95 |                             output_scores=True,
 96 |                         )
 97 | 
 98 |                 sequences = cont.sequences
 99 |                 cur_reponse = self.tokenizer.convert_ids_to_tokens(sequences[0])[0].lower().strip()
100 | 
101 |                 if cur_reponse not in ['yes', 'no']:    break
102 |                 else:   succeed = True
103 | 
104 |                 scores = torch.cat([score.unsqueeze(1) for score in cont.scores], dim=1)
105 |                 scores = torch.nn.functional.softmax(scores, dim=-1)
106 |                 first_token_prob = scores[0, 0] 
107 |                 yes_prob = first_token_prob[self.yes_token_id].item()
108 |                 no_prob = first_token_prob[self.no_token_id].item()
109 |                 # print("==>", cur_reponse, yes_prob, no_prob)
110 |                 # import ipdb; ipdb.set_trace()
111 |                 
112 |             if not succeed:
113 |                 print("Failed to generate a valid 'yes' or 'no' answer after maximum retries. Reponse:" + cur_reponse)
114 |                 # return False, 0.
115 |                 results.append(0)
116 |                 continue
117 |             
118 |             results.append(yes_prob/(yes_prob+no_prob))
119 | 
120 |         return results
121 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/janus/models/clip_encoder.py:
--------------------------------------------------------------------------------
  1 | # Copyright (c) 2023-2024 DeepSeek.
  2 | #
  3 | # Permission is hereby granted, free of charge, to any person obtaining a copy of
  4 | # this software and associated documentation files (the "Software"), to deal in
  5 | # the Software without restriction, including without limitation the rights to
  6 | # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
  7 | # the Software, and to permit persons to whom the Software is furnished to do so,
  8 | # subject to the following conditions:
  9 | #
 10 | # The above copyright notice and this permission notice shall be included in all
 11 | # copies or substantial portions of the Software.
 12 | #
 13 | # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
 14 | # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
 15 | # FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
 16 | # COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
 17 | # IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
 18 | # CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
 19 | 
 20 | from typing import Dict, List, Literal, Optional, Tuple, Union
 21 | 
 22 | import torch
 23 | import torch.nn as nn
 24 | import torchvision.transforms
 25 | from einops import rearrange
 26 | 
 27 | from janus.models.siglip_vit import create_siglip_vit
 28 | 
 29 | 
 30 | class CLIPVisionTower(nn.Module):
 31 |     def __init__(
 32 |         self,
 33 |         model_name: str = "siglip_large_patch16_384",
 34 |         image_size: Union[Tuple[int, int], int] = 336,
 35 |         select_feature: str = "patch",
 36 |         select_layer: int = -2,
 37 |         select_layers: list = None,
 38 |         ckpt_path: str = "",
 39 |         pixel_mean: Optional[List[float]] = None,
 40 |         pixel_std: Optional[List[float]] = None,
 41 |         **kwargs,
 42 |     ):
 43 |         super().__init__()
 44 | 
 45 |         self.model_name = model_name
 46 |         self.select_feature = select_feature
 47 |         self.select_layer = select_layer
 48 |         self.select_layers = select_layers
 49 | 
 50 |         vision_tower_params = {
 51 |             "model_name": model_name,
 52 |             "image_size": image_size,
 53 |             "ckpt_path": ckpt_path,
 54 |             "select_layer": select_layer,
 55 |         }
 56 |         vision_tower_params.update(kwargs)
 57 |         self.vision_tower, self.forward_kwargs = self.build_vision_tower(
 58 |             vision_tower_params
 59 |         )
 60 | 
 61 |         if pixel_mean is not None and pixel_std is not None:
 62 |             image_norm = torchvision.transforms.Normalize(
 63 |                 mean=pixel_mean, std=pixel_std
 64 |             )
 65 |         else:
 66 |             image_norm = None
 67 | 
 68 |         self.image_norm = image_norm
 69 | 
 70 |     def build_vision_tower(self, vision_tower_params):
 71 |         if self.model_name.startswith("siglip"):
 72 |             self.select_feature = "same"
 73 |             vision_tower = create_siglip_vit(**vision_tower_params)
 74 |             forward_kwargs = dict()
 75 | 
 76 |         elif self.model_name.startswith("sam"):
 77 |             vision_tower = create_sam_vit(**vision_tower_params)
 78 |             forward_kwargs = dict()
 79 | 
 80 |         else:  # huggingface
 81 |             from transformers import CLIPVisionModel
 82 | 
 83 |             vision_tower = CLIPVisionModel.from_pretrained(**vision_tower_params)
 84 |             forward_kwargs = dict(output_hidden_states=True)
 85 | 
 86 |         return vision_tower, forward_kwargs
 87 | 
 88 |     def feature_select(self, image_forward_outs):
 89 |         if isinstance(image_forward_outs, torch.Tensor):
 90 |             # the output has been the self.select_layer"s features
 91 |             image_features = image_forward_outs
 92 |         else:
 93 |             image_features = image_forward_outs.hidden_states[self.select_layer]
 94 | 
 95 |         if self.select_feature == "patch":
 96 |             # if the output has cls_token
 97 |             image_features = image_features[:, 1:]
 98 |         elif self.select_feature == "cls_patch":
 99 |             image_features = image_features
100 |         elif self.select_feature == "same":
101 |             image_features = image_features
102 | 
103 |         else:
104 |             raise ValueError(f"Unexpected select feature: {self.select_feature}")
105 |         return image_features
106 | 
107 |     def forward(self, images):
108 |         """
109 | 
110 |         Args:
111 |             images (torch.Tensor): [b, 3, H, W]
112 | 
113 |         Returns:
114 |             image_features (torch.Tensor): [b, n_patch, d]
115 |         """
116 | 
117 |         if self.image_norm is not None:
118 |             images = self.image_norm(images)
119 | 
120 |         image_forward_outs = self.vision_tower(images, **self.forward_kwargs)
121 |         image_features = self.feature_select(image_forward_outs)
122 |         return image_features
123 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/janus/janusflow/models/clip_encoder.py:
--------------------------------------------------------------------------------
  1 | # Copyright (c) 2023-2024 DeepSeek.
  2 | #
  3 | # Permission is hereby granted, free of charge, to any person obtaining a copy of
  4 | # this software and associated documentation files (the "Software"), to deal in
  5 | # the Software without restriction, including without limitation the rights to
  6 | # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of
  7 | # the Software, and to permit persons to whom the Software is furnished to do so,
  8 | # subject to the following conditions:
  9 | #
 10 | # The above copyright notice and this permission notice shall be included in all
 11 | # copies or substantial portions of the Software.
 12 | #
 13 | # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
 14 | # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
 15 | # FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
 16 | # COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
 17 | # IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
 18 | # CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
 19 | 
 20 | from typing import Dict, List, Literal, Optional, Tuple, Union
 21 | 
 22 | import torch
 23 | import torch.nn as nn
 24 | import torchvision.transforms
 25 | from einops import rearrange
 26 | 
 27 | from janus.janusflow.models.siglip_vit import create_siglip_vit
 28 | 
 29 | 
 30 | class CLIPVisionTower(nn.Module):
 31 |     def __init__(
 32 |         self,
 33 |         model_name: str = "siglip_large_patch16_384",
 34 |         image_size: Union[Tuple[int, int], int] = 336,
 35 |         select_feature: str = "patch",
 36 |         select_layer: int = -2,
 37 |         select_layers: list = None,
 38 |         ckpt_path: str = "",
 39 |         pixel_mean: Optional[List[float]] = None,
 40 |         pixel_std: Optional[List[float]] = None,
 41 |         **kwargs,
 42 |     ):
 43 |         super().__init__()
 44 | 
 45 |         self.model_name = model_name
 46 |         self.select_feature = select_feature
 47 |         self.select_layer = select_layer
 48 |         self.select_layers = select_layers
 49 | 
 50 |         vision_tower_params = {
 51 |             "model_name": model_name,
 52 |             "image_size": image_size,
 53 |             "ckpt_path": ckpt_path,
 54 |             "select_layer": select_layer,
 55 |         }
 56 |         vision_tower_params.update(kwargs)
 57 |         self.vision_tower, self.forward_kwargs = self.build_vision_tower(
 58 |             vision_tower_params
 59 |         )
 60 | 
 61 |         if pixel_mean is not None and pixel_std is not None:
 62 |             image_norm = torchvision.transforms.Normalize(
 63 |                 mean=pixel_mean, std=pixel_std
 64 |             )
 65 |         else:
 66 |             image_norm = None
 67 | 
 68 |         self.image_norm = image_norm
 69 | 
 70 |     def build_vision_tower(self, vision_tower_params):
 71 |         if self.model_name.startswith("siglip"):
 72 |             self.select_feature = "same"
 73 |             vision_tower = create_siglip_vit(**vision_tower_params)
 74 |             forward_kwargs = dict()
 75 | 
 76 |         elif self.model_name.startswith("sam"):
 77 |             vision_tower = create_sam_vit(**vision_tower_params)
 78 |             forward_kwargs = dict()
 79 | 
 80 |         else:  # huggingface
 81 |             from transformers import CLIPVisionModel
 82 | 
 83 |             vision_tower = CLIPVisionModel.from_pretrained(**vision_tower_params)
 84 |             forward_kwargs = dict(output_hidden_states=True)
 85 | 
 86 |         return vision_tower, forward_kwargs
 87 | 
 88 |     def feature_select(self, image_forward_outs):
 89 |         if isinstance(image_forward_outs, torch.Tensor):
 90 |             # the output has been the self.select_layer"s features
 91 |             image_features = image_forward_outs
 92 |         else:
 93 |             image_features = image_forward_outs.hidden_states[self.select_layer]
 94 | 
 95 |         if self.select_feature == "patch":
 96 |             # if the output has cls_token
 97 |             image_features = image_features[:, 1:]
 98 |         elif self.select_feature == "cls_patch":
 99 |             image_features = image_features
100 |         elif self.select_feature == "same":
101 |             image_features = image_features
102 | 
103 |         else:
104 |             raise ValueError(f"Unexpected select feature: {self.select_feature}")
105 |         return image_features
106 | 
107 |     def forward(self, images):
108 |         """
109 | 
110 |         Args:
111 |             images (torch.Tensor): [b, 3, H, W]
112 | 
113 |         Returns:
114 |             image_features (torch.Tensor): [b, n_patch, d]
115 |         """
116 | 
117 |         if self.image_norm is not None:
118 |             images = self.image_norm(images)
119 | 
120 |         image_forward_outs = self.vision_tower(images, **self.forward_kwargs)
121 |         image_features = self.feature_select(image_forward_outs)
122 |         return image_features
123 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/dev_eva_clip/eva_clip/timm_model.py:
--------------------------------------------------------------------------------
  1 | """ timm model adapter
  2 | 
  3 | Wraps timm (https://github.com/rwightman/pytorch-image-models) models for use as a vision tower in CLIP model.
  4 | """
  5 | 
  6 | import logging
  7 | from collections import OrderedDict
  8 | 
  9 | import torch
 10 | import torch.nn as nn
 11 | 
 12 | try:
 13 |     import timm
 14 |     from timm.models.layers import Mlp, to_2tuple
 15 | 
 16 |     try:
 17 |         # old timm imports < 0.8.1
 18 |         from timm.models.layers.attention_pool2d import RotAttentionPool2d
 19 |         from timm.models.layers.attention_pool2d import AttentionPool2d as AbsAttentionPool2d
 20 |     except ImportError:
 21 |         # new timm imports >= 0.8.1
 22 |         from timm.layers import RotAttentionPool2d
 23 |         from timm.layers import AttentionPool2d as AbsAttentionPool2d
 24 | except ImportError:
 25 |     timm = None
 26 | 
 27 | from .utils import freeze_batch_norm_2d
 28 | 
 29 | 
 30 | class TimmModel(nn.Module):
 31 |     """timm model adapter
 32 |     # FIXME this adapter is a work in progress, may change in ways that break weight compat
 33 |     """
 34 | 
 35 |     def __init__(self, model_name, embed_dim, image_size=224, pool="avg", proj="linear", proj_bias=False, drop=0.0, pretrained=False):
 36 |         super().__init__()
 37 |         if timm is None:
 38 |             raise RuntimeError("Please `pip install timm` to use timm models.")
 39 | 
 40 |         self.image_size = to_2tuple(image_size)
 41 |         self.trunk = timm.create_model(model_name, pretrained=pretrained)
 42 |         feat_size = self.trunk.default_cfg.get("pool_size", None)
 43 |         feature_ndim = 1 if not feat_size else 2
 44 |         if pool in ("abs_attn", "rot_attn"):
 45 |             assert feature_ndim == 2
 46 |             # if attn pooling used, remove both classifier and default pool
 47 |             self.trunk.reset_classifier(0, global_pool="")
 48 |         else:
 49 |             # reset global pool if pool config set, otherwise leave as network default
 50 |             reset_kwargs = dict(global_pool=pool) if pool else {}
 51 |             self.trunk.reset_classifier(0, **reset_kwargs)
 52 |         prev_chs = self.trunk.num_features
 53 | 
 54 |         head_layers = OrderedDict()
 55 |         if pool == "abs_attn":
 56 |             head_layers["pool"] = AbsAttentionPool2d(prev_chs, feat_size=feat_size, out_features=embed_dim)
 57 |             prev_chs = embed_dim
 58 |         elif pool == "rot_attn":
 59 |             head_layers["pool"] = RotAttentionPool2d(prev_chs, out_features=embed_dim)
 60 |             prev_chs = embed_dim
 61 |         else:
 62 |             assert proj, "projection layer needed if non-attention pooling is used."
 63 | 
 64 |         # NOTE attention pool ends with a projection layer, so proj should usually be set to '' if such pooling is used
 65 |         if proj == "linear":
 66 |             head_layers["drop"] = nn.Dropout(drop)
 67 |             head_layers["proj"] = nn.Linear(prev_chs, embed_dim, bias=proj_bias)
 68 |         elif proj == "mlp":
 69 |             head_layers["mlp"] = Mlp(prev_chs, 2 * embed_dim, embed_dim, drop=drop, bias=(True, proj_bias))
 70 | 
 71 |         self.head = nn.Sequential(head_layers)
 72 | 
 73 |     def lock(self, unlocked_groups=0, freeze_bn_stats=False):
 74 |         """lock modules
 75 |         Args:
 76 |             unlocked_groups (int): leave last n layer groups unlocked (default: 0)
 77 |         """
 78 |         if not unlocked_groups:
 79 |             # lock full model
 80 |             for param in self.trunk.parameters():
 81 |                 param.requires_grad = False
 82 |             if freeze_bn_stats:
 83 |                 freeze_batch_norm_2d(self.trunk)
 84 |         else:
 85 |             # NOTE: partial freeze requires latest timm (master) branch and is subject to change
 86 |             try:
 87 |                 # FIXME import here until API stable and in an official release
 88 |                 from timm.models.helpers import group_parameters, group_modules
 89 |             except ImportError:
 90 |                 raise RuntimeError("Please install latest timm `pip install git+https://github.com/rwightman/pytorch-image-models`")
 91 |             matcher = self.trunk.group_matcher()
 92 |             gparams = group_parameters(self.trunk, matcher)
 93 |             max_layer_id = max(gparams.keys())
 94 |             max_layer_id = max_layer_id - unlocked_groups
 95 |             for group_idx in range(max_layer_id + 1):
 96 |                 group = gparams[group_idx]
 97 |                 for param in group:
 98 |                     self.trunk.get_parameter(param).requires_grad = False
 99 |             if freeze_bn_stats:
100 |                 gmodules = group_modules(self.trunk, matcher, reverse=True)
101 |                 gmodules = {k for k, v in gmodules.items() if v <= max_layer_id}
102 |                 freeze_batch_norm_2d(self.trunk, gmodules)
103 | 
104 |     @torch.jit.ignore
105 |     def set_grad_checkpointing(self, enable=True):
106 |         try:
107 |             self.trunk.set_grad_checkpointing(enable)
108 |         except Exception as e:
109 |             logging.warning("grad checkpointing not supported for this timm image tower, continuing without...")
110 | 
111 |     def forward(self, x):
112 |         x = self.trunk(x)
113 |         x = self.head(x)
114 |         return x
115 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/language_model/llava_gemma.py:
--------------------------------------------------------------------------------
  1 | #    Copyright 2024 Duc Q. Nguyen, Haotian Liu and Bo Li
  2 | #
  3 | #    Licensed under the Apache License, Version 2.0 (the "License");
  4 | #    you may not use this file except in compliance with the License.
  5 | #    You may obtain a copy of the License at
  6 | #
  7 | #        http://www.apache.org/licenses/LICENSE-2.0
  8 | #
  9 | #    Unless required by applicable law or agreed to in writing, software
 10 | #    distributed under the License is distributed on an "AS IS" BASIS,
 11 | #    WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
 12 | #    See the License for the specific language governing permissions and
 13 | #    limitations under the License.
 14 | 
 15 | 
 16 | from typing import List, Optional, Tuple, Union
 17 | 
 18 | import torch
 19 | import torch.nn as nn
 20 | from torch.nn import CrossEntropyLoss
 21 | 
 22 | from transformers import AutoConfig, AutoModelForCausalLM, GemmaConfig, GemmaModel, GemmaForCausalLM
 23 | 
 24 | from transformers.modeling_outputs import CausalLMOutputWithPast
 25 | from transformers.generation.utils import GenerateOutput
 26 | 
 27 | from ..llava_arch import LlavaMetaModel, LlavaMetaForCausalLM
 28 | 
 29 | 
 30 | class LlavaGemmaConfig(GemmaConfig):
 31 |     model_type = "llava_gemma"
 32 | 
 33 | 
 34 | class LlavaGemmaModel(LlavaMetaModel, GemmaModel):
 35 |     config_class = LlavaGemmaConfig
 36 | 
 37 |     def __init__(self, config: GemmaConfig):
 38 |         super(LlavaGemmaModel, self).__init__(config)
 39 | 
 40 | 
 41 | class LlavaGemmaForCausalLM(GemmaForCausalLM, LlavaMetaForCausalLM):
 42 |     config_class = LlavaGemmaConfig
 43 | 
 44 |     def __init__(self, config):
 45 |         super(GemmaForCausalLM, self).__init__(config)
 46 |         self.model = LlavaGemmaModel(config)
 47 | 
 48 |         self.lm_head = nn.Linear(config.hidden_size, config.vocab_size, bias=False)
 49 | 
 50 |         # Initialize weights and apply final processing
 51 |         self.post_init()
 52 | 
 53 |     def get_model(self):
 54 |         return self.model
 55 | 
 56 |     def forward(
 57 |         self,
 58 |         input_ids: torch.LongTensor = None,
 59 |         attention_mask: Optional[torch.Tensor] = None,
 60 |         position_ids: Optional[torch.LongTensor] = None,
 61 |         past_key_values: Optional[List[torch.FloatTensor]] = None,
 62 |         inputs_embeds: Optional[torch.FloatTensor] = None,
 63 |         labels: Optional[torch.LongTensor] = None,
 64 |         use_cache: Optional[bool] = None,
 65 |         output_attentions: Optional[bool] = None,
 66 |         output_hidden_states: Optional[bool] = None,
 67 |         images: Optional[torch.FloatTensor] = None,
 68 |         image_sizes: Optional[List[List[int]]] = None,
 69 |         return_dict: Optional[bool] = None,
 70 |         cache_position: Optional[torch.LongTensor] = None,
 71 |     ) -> Union[Tuple, CausalLMOutputWithPast]:
 72 | 
 73 |         if inputs_embeds is None:
 74 |             (input_ids, position_ids, attention_mask, past_key_values, inputs_embeds, labels) = self.prepare_inputs_labels_for_multimodal(input_ids, position_ids, attention_mask, past_key_values, labels, images, image_sizes)
 75 | 
 76 |         return super().forward(
 77 |             input_ids=input_ids,
 78 |             attention_mask=attention_mask,
 79 |             position_ids=position_ids,
 80 |             past_key_values=past_key_values,
 81 |             inputs_embeds=inputs_embeds,
 82 |             labels=labels,
 83 |             use_cache=use_cache,
 84 |             output_attentions=output_attentions,
 85 |             output_hidden_states=output_hidden_states,
 86 |             return_dict=return_dict,
 87 |             cache_position=cache_position,
 88 |         )
 89 | 
 90 |     @torch.no_grad()
 91 |     def generate(
 92 |         self,
 93 |         inputs: Optional[torch.Tensor] = None,
 94 |         images: Optional[torch.Tensor] = None,
 95 |         image_sizes: Optional[torch.Tensor] = None,
 96 |         **kwargs,
 97 |     ) -> Union[GenerateOutput, torch.LongTensor]:
 98 |         position_ids = kwargs.pop("position_ids", None)
 99 |         attention_mask = kwargs.pop("attention_mask", None)
100 |         if "inputs_embeds" in kwargs:
101 |             raise NotImplementedError("`inputs_embeds` is not supported")
102 | 
103 |         if images is not None:
104 |             (inputs, position_ids, attention_mask, _, inputs_embeds, _) = self.prepare_inputs_labels_for_multimodal(inputs, position_ids, attention_mask, None, None, images, image_sizes=image_sizes)
105 |         else:
106 |             inputs_embeds = self.get_model().embed_tokens(inputs)
107 | 
108 |         return super().generate(position_ids=position_ids, attention_mask=attention_mask, inputs_embeds=inputs_embeds, **kwargs)
109 | 
110 |     def prepare_inputs_for_generation(self, input_ids, past_key_values=None, inputs_embeds=None, **kwargs):
111 |         images = kwargs.pop("images", None)
112 |         image_sizes = kwargs.pop("image_sizes", None)
113 |         inputs = super().prepare_inputs_for_generation(input_ids, past_key_values=past_key_values, inputs_embeds=inputs_embeds, **kwargs)
114 |         if images is not None:
115 |             inputs["images"] = images
116 |         if image_sizes is not None:
117 |             inputs["image_sizes"] = image_sizes
118 |         return inputs
119 | 
120 | 
121 | AutoConfig.register("llava_gemma", LlavaGemmaConfig)
122 | AutoModelForCausalLM.register(LlavaGemmaConfig, LlavaGemmaForCausalLM)
123 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_resampler/perceiver.py:
--------------------------------------------------------------------------------
  1 | """
  2 | Taken from https://github.com/lucidrains/flamingo-pytorch
  3 | """
  4 | 
  5 | import torch
  6 | from einops import rearrange, repeat
  7 | 
  8 | try:
  9 |     from einops_exts import rearrange_many
 10 | except:
 11 |     pass
 12 | 
 13 | from torch import einsum, nn
 14 | 
 15 | 
 16 | def exists(val):
 17 |     return val is not None
 18 | 
 19 | 
 20 | def FeedForward(dim, mult=4):
 21 |     inner_dim = int(dim * mult)
 22 |     return nn.Sequential(
 23 |         nn.LayerNorm(dim),
 24 |         nn.Linear(dim, inner_dim, bias=False),
 25 |         nn.GELU(),
 26 |         nn.Linear(inner_dim, dim, bias=False),
 27 |     )
 28 | 
 29 | 
 30 | class PerceiverAttention(nn.Module):
 31 |     def __init__(self, *, dim, dim_head=64, heads=8):
 32 |         super().__init__()
 33 |         self.scale = dim_head**-0.5
 34 |         self.heads = heads
 35 |         inner_dim = dim_head * heads
 36 | 
 37 |         self.norm_media = nn.LayerNorm(dim)
 38 |         self.norm_latents = nn.LayerNorm(dim)
 39 | 
 40 |         self.to_q = nn.Linear(dim, inner_dim, bias=False)
 41 |         self.to_kv = nn.Linear(dim, inner_dim * 2, bias=False)
 42 |         self.to_out = nn.Linear(inner_dim, dim, bias=False)
 43 | 
 44 |     def forward(self, x, latents):
 45 |         """
 46 |         Args:
 47 |             x (torch.Tensor): image features
 48 |                 shape (b, T, n1, D)
 49 |             latent (torch.Tensor): latent features
 50 |                 shape (b, T, n2, D)
 51 |         """
 52 |         x = self.norm_media(x)
 53 |         latents = self.norm_latents(latents)
 54 | 
 55 |         h = self.heads
 56 | 
 57 |         q = self.to_q(latents)
 58 |         kv_input = torch.cat((x, latents), dim=-2)
 59 |         k, v = self.to_kv(kv_input).chunk(2, dim=-1)
 60 |         q, k, v = rearrange_many((q, k, v), "b t n (h d) -> b h t n d", h=h)
 61 |         q = q * self.scale
 62 | 
 63 |         # attention
 64 |         sim = einsum("... i d, ... j d  -> ... i j", q, k)
 65 |         sim = sim - sim.amax(dim=-1, keepdim=True).detach()
 66 |         attn = sim.softmax(dim=-1)
 67 | 
 68 |         out = einsum("... i j, ... j d -> ... i d", attn, v)
 69 |         out = rearrange(out, "b h t n d -> b t n (h d)", h=h)
 70 |         return self.to_out(out)
 71 | 
 72 | 
 73 | class PerceiverResamplerModule(nn.Module):
 74 |     def __init__(
 75 |         self,
 76 |         *,
 77 |         dim,
 78 |         depth=6,
 79 |         dim_head=64,
 80 |         heads=8,
 81 |         num_latents=64,
 82 |         max_num_media=None,
 83 |         max_num_frames=None,
 84 |         ff_mult=4,
 85 |     ):
 86 |         super().__init__()
 87 |         self.latents = nn.Parameter(torch.randn(num_latents, dim))
 88 |         self.frame_embs = nn.Parameter(torch.randn(max_num_frames, dim)) if exists(max_num_frames) else None
 89 |         self.media_time_embs = nn.Parameter(torch.randn(max_num_media, 1, dim)) if exists(max_num_media) else None
 90 | 
 91 |         self.layers = nn.ModuleList([])
 92 |         for _ in range(depth):
 93 |             self.layers.append(
 94 |                 nn.ModuleList(
 95 |                     [
 96 |                         PerceiverAttention(dim=dim, dim_head=dim_head, heads=heads),
 97 |                         FeedForward(dim=dim, mult=ff_mult) if ff_mult > 0 else nn.Identity(),
 98 |                     ]
 99 |                 )
100 |             )
101 | 
102 |         self.norm = nn.LayerNorm(dim)
103 | 
104 |     def forward(self, x):
105 |         """
106 |         Args:
107 |             x (torch.Tensor): image features
108 |                 shape (b, T, F, v, D)
109 |         Returns:
110 |             shape (b, T, n, D) where n is self.num_latents
111 |         """
112 |         b, T, F, v = x.shape[:4]
113 | 
114 |         # frame and media time embeddings
115 |         if exists(self.frame_embs):
116 |             frame_embs = repeat(self.frame_embs[:F], "F d -> b T F v d", b=b, T=T, v=v)
117 |             x = x + frame_embs
118 |         x = rearrange(x, "b T F v d -> b T (F v) d")  # flatten the frame and spatial dimensions
119 |         if exists(self.media_time_embs):
120 |             x = x + self.media_time_embs[:T]
121 | 
122 |         # blocks
123 |         latents = repeat(self.latents, "n d -> b T n d", b=b, T=T)
124 |         for attn, ff in self.layers:
125 |             latents = attn(x, latents) + latents
126 |             latents = ff(latents) + latents
127 |         return self.norm(latents)
128 | 
129 | 
130 | class PerceiverResampler(nn.Module):
131 |     def __init__(self, model_args, vision_tower):
132 |         super().__init__()
133 | 
134 |         self.depth = model_args.mm_perceiver_depth
135 |         self.num_latents = model_args.mm_perceiver_latents
136 |         self.ff_mult = model_args.mm_perceiver_ff_mult
137 |         self.pretrained = model_args.mm_perceiver_pretrained
138 | 
139 |         self.perceiver = PerceiverResamplerModule(dim=vision_tower.hidden_size, depth=self.depth, num_latents=self.num_latents, ff_mult=self.ff_mult)
140 | 
141 |         if self.pretrained is not None:
142 |             self.load_state_dict(torch.load(self.pretrained))
143 | 
144 |     def forward(self, image_features, *args, **kwargs):
145 |         return self.perceiver(image_features[:, None, None]).squeeze(1)
146 | 
147 |     @property
148 |     def config(self):
149 |         return {
150 |             "mm_resampler_type": "perceiver",
151 |             "mm_perceiver_depth": self.depth,
152 |             "mm_perceiver_latents": self.num_latents,
153 |             "mm_perceiver_ff_mult": self.ff_mult,
154 |             "mm_perceiver_pretrained": self.pretrained,
155 |         }
156 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/language_model/llava_mistral.py:
--------------------------------------------------------------------------------
  1 | #    Copyright 2023 Haotian Liu
  2 | #
  3 | #    Licensed under the Apache License, Version 2.0 (the "License");
  4 | #    you may not use this file except in compliance with the License.
  5 | #    You may obtain a copy of the License at
  6 | #
  7 | #        http://www.apache.org/licenses/LICENSE-2.0
  8 | #
  9 | #    Unless required by applicable law or agreed to in writing, software
 10 | #    distributed under the License is distributed on an "AS IS" BASIS,
 11 | #    WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
 12 | #    See the License for the specific language governing permissions and
 13 | #    limitations under the License.
 14 | 
 15 | 
 16 | from typing import List, Optional, Tuple, Union
 17 | 
 18 | import torch
 19 | import torch.nn as nn
 20 | from torch.nn import CrossEntropyLoss
 21 | 
 22 | from transformers import AutoConfig, AutoModelForCausalLM, MistralConfig, MistralModel, MistralForCausalLM, GenerationConfig
 23 | 
 24 | from transformers.modeling_outputs import CausalLMOutputWithPast
 25 | from transformers.generation.utils import GenerateOutput
 26 | 
 27 | from ..llava_arch import LlavaMetaModel, LlavaMetaForCausalLM
 28 | 
 29 | 
 30 | class LlavaMistralConfig(MistralConfig):
 31 |     model_type = "llava_mistral"
 32 |     temperature: float = 0.0  # reset to 0.0, previously 0.9 for Vicuna
 33 |     max_new_tokens: int = 1024
 34 |     do_sample: bool = False
 35 |     top_p: Optional[float] = None
 36 | 
 37 | 
 38 | class LlavaMistralModel(LlavaMetaModel, MistralModel):
 39 |     config_class = LlavaMistralConfig
 40 | 
 41 |     def __init__(self, config: MistralConfig):
 42 |         super(LlavaMistralModel, self).__init__(config)
 43 | 
 44 | 
 45 | class LlavaMistralForCausalLM(MistralForCausalLM, LlavaMetaForCausalLM):
 46 |     config_class = LlavaMistralConfig
 47 | 
 48 |     def __init__(self, config):
 49 |         super(MistralForCausalLM, self).__init__(config)
 50 | 
 51 |         config.model_type = "llava_mistral"
 52 |         config.rope_scaling = None
 53 | 
 54 |         self.model = LlavaMistralModel(config)
 55 |         self.lm_head = nn.Linear(config.hidden_size, config.vocab_size, bias=False)
 56 |         # Initialize weights and apply final processing
 57 |         self.post_init()
 58 | 
 59 |     def get_model(self):
 60 |         return self.model
 61 | 
 62 |     def forward(
 63 |         self,
 64 |         input_ids: torch.LongTensor = None,
 65 |         attention_mask: Optional[torch.Tensor] = None,
 66 |         position_ids: Optional[torch.LongTensor] = None,
 67 |         past_key_values: Optional[List[torch.FloatTensor]] = None,
 68 |         inputs_embeds: Optional[torch.FloatTensor] = None,
 69 |         labels: Optional[torch.LongTensor] = None,
 70 |         use_cache: Optional[bool] = None,
 71 |         output_attentions: Optional[bool] = None,
 72 |         output_hidden_states: Optional[bool] = None,
 73 |         images: Optional[torch.FloatTensor] = None,
 74 |         image_sizes: Optional[List[List[int]]] = None,
 75 |         return_dict: Optional[bool] = None,
 76 |         cache_position=None,
 77 |     ) -> Union[Tuple, CausalLMOutputWithPast]:
 78 | 
 79 |         if inputs_embeds is None:
 80 |             (input_ids, position_ids, attention_mask, past_key_values, inputs_embeds, labels) = self.prepare_inputs_labels_for_multimodal(input_ids, position_ids, attention_mask, past_key_values, labels, images, image_sizes)
 81 | 
 82 |         return super().forward(
 83 |             input_ids=input_ids,
 84 |             attention_mask=attention_mask,
 85 |             position_ids=position_ids,
 86 |             past_key_values=past_key_values,
 87 |             inputs_embeds=inputs_embeds,
 88 |             labels=labels,
 89 |             use_cache=use_cache,
 90 |             output_attentions=output_attentions,
 91 |             output_hidden_states=output_hidden_states,
 92 |             return_dict=return_dict,
 93 |         )
 94 | 
 95 |     @torch.no_grad()
 96 |     def generate(
 97 |         self,
 98 |         inputs: Optional[torch.Tensor] = None,
 99 |         images: Optional[torch.Tensor] = None,
100 |         image_sizes: Optional[torch.Tensor] = None,
101 |         **kwargs,
102 |     ) -> Union[GenerateOutput, torch.LongTensor]:
103 |         position_ids = kwargs.pop("position_ids", None)
104 |         attention_mask = kwargs.pop("attention_mask", None)
105 |         if "inputs_embeds" in kwargs:
106 |             raise NotImplementedError("`inputs_embeds` is not supported")
107 | 
108 |         if images is not None:
109 |             (inputs, position_ids, attention_mask, _, inputs_embeds, _) = self.prepare_inputs_labels_for_multimodal(inputs, position_ids, attention_mask, None, None, images, image_sizes=image_sizes)
110 |         else:
111 |             inputs_embeds = self.get_model().embed_tokens(inputs)
112 | 
113 |         return super().generate(position_ids=position_ids, attention_mask=attention_mask, inputs_embeds=inputs_embeds, **kwargs)
114 | 
115 |     def prepare_inputs_for_generation(self, input_ids, past_key_values=None, inputs_embeds=None, **kwargs):
116 |         images = kwargs.pop("images", None)
117 |         image_sizes = kwargs.pop("image_sizes", None)
118 |         inputs = super().prepare_inputs_for_generation(input_ids, past_key_values=past_key_values, inputs_embeds=inputs_embeds, **kwargs)
119 |         if images is not None:
120 |             inputs["images"] = images
121 |         if image_sizes is not None:
122 |             inputs["image_sizes"] = image_sizes
123 |         return inputs
124 | 
125 | 
126 | AutoConfig.register("llava_mistral", LlavaMistralConfig)
127 | AutoModelForCausalLM.register(LlavaMistralConfig, LlavaMistralForCausalLM)
128 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/dev_eva_clip/eva_clip/rope.py:
--------------------------------------------------------------------------------
  1 | from math import pi
  2 | import torch
  3 | from torch import nn
  4 | from einops import rearrange, repeat
  5 | import logging
  6 | 
  7 | 
  8 | def broadcat(tensors, dim=-1):
  9 |     num_tensors = len(tensors)
 10 |     shape_lens = set(list(map(lambda t: len(t.shape), tensors)))
 11 |     assert len(shape_lens) == 1, "tensors must all have the same number of dimensions"
 12 |     shape_len = list(shape_lens)[0]
 13 |     dim = (dim + shape_len) if dim < 0 else dim
 14 |     dims = list(zip(*map(lambda t: list(t.shape), tensors)))
 15 |     expandable_dims = [(i, val) for i, val in enumerate(dims) if i != dim]
 16 |     assert all([*map(lambda t: len(set(t[1])) <= 2, expandable_dims)]), "invalid dimensions for broadcastable concatentation"
 17 |     max_dims = list(map(lambda t: (t[0], max(t[1])), expandable_dims))
 18 |     expanded_dims = list(map(lambda t: (t[0], (t[1],) * num_tensors), max_dims))
 19 |     expanded_dims.insert(dim, (dim, dims[dim]))
 20 |     expandable_shapes = list(zip(*map(lambda t: t[1], expanded_dims)))
 21 |     tensors = list(map(lambda t: t[0].expand(*t[1]), zip(tensors, expandable_shapes)))
 22 |     return torch.cat(tensors, dim=dim)
 23 | 
 24 | 
 25 | def rotate_half(x):
 26 |     x = rearrange(x, "... (d r) -> ... d r", r=2)
 27 |     x1, x2 = x.unbind(dim=-1)
 28 |     x = torch.stack((-x2, x1), dim=-1)
 29 |     return rearrange(x, "... d r -> ... (d r)")
 30 | 
 31 | 
 32 | class VisionRotaryEmbedding(nn.Module):
 33 |     def __init__(
 34 |         self,
 35 |         dim,
 36 |         pt_seq_len,
 37 |         ft_seq_len=None,
 38 |         custom_freqs=None,
 39 |         freqs_for="lang",
 40 |         theta=10000,
 41 |         max_freq=10,
 42 |         num_freqs=1,
 43 |     ):
 44 |         super().__init__()
 45 |         if custom_freqs:
 46 |             freqs = custom_freqs
 47 |         elif freqs_for == "lang":
 48 |             freqs = 1.0 / (theta ** (torch.arange(0, dim, 2)[: (dim // 2)].float() / dim))
 49 |         elif freqs_for == "pixel":
 50 |             freqs = torch.linspace(1.0, max_freq / 2, dim // 2) * pi
 51 |         elif freqs_for == "constant":
 52 |             freqs = torch.ones(num_freqs).float()
 53 |         else:
 54 |             raise ValueError(f"unknown modality {freqs_for}")
 55 | 
 56 |         if ft_seq_len is None:
 57 |             ft_seq_len = pt_seq_len
 58 |         t = torch.arange(ft_seq_len) / ft_seq_len * pt_seq_len
 59 | 
 60 |         freqs_h = torch.einsum("..., f -> ... f", t, freqs)
 61 |         freqs_h = repeat(freqs_h, "... n -> ... (n r)", r=2)
 62 | 
 63 |         freqs_w = torch.einsum("..., f -> ... f", t, freqs)
 64 |         freqs_w = repeat(freqs_w, "... n -> ... (n r)", r=2)
 65 | 
 66 |         freqs = broadcat((freqs_h[:, None, :], freqs_w[None, :, :]), dim=-1)
 67 | 
 68 |         self.register_buffer("freqs_cos", freqs.cos())
 69 |         self.register_buffer("freqs_sin", freqs.sin())
 70 | 
 71 |         logging.info(f"Shape of rope freq: {self.freqs_cos.shape}")
 72 | 
 73 |     def forward(self, t, start_index=0):
 74 |         rot_dim = self.freqs_cos.shape[-1]
 75 |         end_index = start_index + rot_dim
 76 |         assert rot_dim <= t.shape[-1], f"feature dimension {t.shape[-1]} is not of sufficient size to rotate in all the positions {rot_dim}"
 77 |         t_left, t, t_right = t[..., :start_index], t[..., start_index:end_index], t[..., end_index:]
 78 |         t = (t * self.freqs_cos) + (rotate_half(t) * self.freqs_sin)
 79 | 
 80 |         return torch.cat((t_left, t, t_right), dim=-1)
 81 | 
 82 | 
 83 | class VisionRotaryEmbeddingFast(nn.Module):
 84 |     def __init__(self, dim, pt_seq_len, ft_seq_len=None, custom_freqs=None, freqs_for="lang", theta=10000, max_freq=10, num_freqs=1, patch_dropout=0.0):
 85 |         super().__init__()
 86 |         if custom_freqs:
 87 |             freqs = custom_freqs
 88 |         elif freqs_for == "lang":
 89 |             freqs = 1.0 / (theta ** (torch.arange(0, dim, 2)[: (dim // 2)].float() / dim))
 90 |         elif freqs_for == "pixel":
 91 |             freqs = torch.linspace(1.0, max_freq / 2, dim // 2) * pi
 92 |         elif freqs_for == "constant":
 93 |             freqs = torch.ones(num_freqs).float()
 94 |         else:
 95 |             raise ValueError(f"unknown modality {freqs_for}")
 96 | 
 97 |         if ft_seq_len is None:
 98 |             ft_seq_len = pt_seq_len
 99 |         t = torch.arange(ft_seq_len) / ft_seq_len * pt_seq_len
100 | 
101 |         freqs = torch.einsum("..., f -> ... f", t, freqs)
102 |         freqs = repeat(freqs, "... n -> ... (n r)", r=2)
103 |         freqs = broadcat((freqs[:, None, :], freqs[None, :, :]), dim=-1)
104 | 
105 |         freqs_cos = freqs.cos().view(-1, freqs.shape[-1])
106 |         freqs_sin = freqs.sin().view(-1, freqs.shape[-1])
107 | 
108 |         self.patch_dropout = patch_dropout
109 | 
110 |         self.register_buffer("freqs_cos", freqs_cos)
111 |         self.register_buffer("freqs_sin", freqs_sin)
112 | 
113 |         logging.info(f"Shape of rope freq: {self.freqs_cos.shape}")
114 | 
115 |     def forward(self, t, patch_indices_keep=None):
116 |         if patch_indices_keep is not None:
117 |             batch = t.size()[0]
118 |             batch_indices = torch.arange(batch)
119 |             batch_indices = batch_indices[..., None]
120 | 
121 |             freqs_cos = repeat(self.freqs_cos, "i j -> n i m j", n=t.shape[0], m=t.shape[1])
122 |             freqs_sin = repeat(self.freqs_sin, "i j -> n i m j", n=t.shape[0], m=t.shape[1])
123 | 
124 |             freqs_cos = freqs_cos[batch_indices, patch_indices_keep]
125 |             freqs_cos = rearrange(freqs_cos, "n i m j -> n m i j")
126 |             freqs_sin = freqs_sin[batch_indices, patch_indices_keep]
127 |             freqs_sin = rearrange(freqs_sin, "n i m j -> n m i j")
128 | 
129 |             return t * freqs_cos + rotate_half(t) * freqs_sin
130 | 
131 |         return t * self.freqs_cos + rotate_half(t) * self.freqs_sin
132 | 


--------------------------------------------------------------------------------
/src/t2i-r1/src/utils/LLaVA-NeXT/llava/model/multimodal_encoder/dev_eva_clip/eva_clip/openai.py:
--------------------------------------------------------------------------------
  1 | """ OpenAI pretrained model functions
  2 | 
  3 | Adapted from https://github.com/openai/CLIP. Originally MIT License, Copyright (c) 2021 OpenAI.
  4 | """
  5 | 
  6 | import os
  7 | import warnings
  8 | from typing import List, Optional, Union
  9 | 
 10 | import torch
 11 | 
 12 | from .model import build_model_from_openai_state_dict, convert_weights_to_lp, get_cast_dtype
 13 | from .pretrained import get_pretrained_url, list_pretrained_models_by_tag, download_pretrained_from_url
 14 | 
 15 | __all__ = ["list_openai_models", "load_openai_model"]
 16 | 
 17 | 
 18 | def list_openai_models() -> List[str]:
 19 |     """Returns the names of available CLIP models"""
 20 |     return list_pretrained_models_by_tag("openai")
 21 | 
 22 | 
 23 | def load_openai_model(
 24 |     name: str,
 25 |     precision: Optional[str] = None,
 26 |     device: Optional[Union[str, torch.device]] = None,
 27 |     jit: bool = True,
 28 |     cache_dir: Optional[str] = None,
 29 | ):
 30 |     """Load a CLIP model
 31 | 
 32 |     Parameters
 33 |     ----------
 34 |     name : str
 35 |         A model name listed by `clip.available_models()`, or the path to a model checkpoint containing the state_dict
 36 |     precision: str
 37 |         Model precision, if None defaults to 'fp32' if device == 'cpu' else 'fp16'.
 38 |     device : Union[str, torch.device]
 39 |         The device to put the loaded model
 40 |     jit : bool
 41 |         Whether to load the optimized JIT model (default) or more hackable non-JIT model.
 42 |     cache_dir : Optional[str]
 43 |         The directory to cache the downloaded model weights
 44 | 
 45 |     Returns
 46 |     -------
 47 |     model : torch.nn.Module
 48 |         The CLIP model
 49 |     preprocess : Callable[[PIL.Image], torch.Tensor]
 50 |         A torchvision transform that converts a PIL image into a tensor that the returned model can take as its input
 51 |     """
 52 |     if device is None:
 53 |         device = "cuda" if torch.cuda.is_available() else "cpu"
 54 |     if precision is None:
 55 |         precision = "fp32" if device == "cpu" else "fp16"
 56 | 
 57 |     if get_pretrained_url(name, "openai"):
 58 |         model_path = download_pretrained_from_url(get_pretrained_url(name, "openai"), cache_dir=cache_dir)
 59 |     elif os.path.isfile(name):
 60 |         model_path = name
 61 |     else:
 62 |         raise RuntimeError(f"Model {name} not found; available models = {list_openai_models()}")
 63 | 
 64 |     try:
 65 |         # loading JIT archive
 66 |         model = torch.jit.load(model_path, map_location=device if jit else "cpu").eval()
 67 |         state_dict = None
 68 |     except RuntimeError:
 69 |         # loading saved state dict
 70 |         if jit:
 71 |             warnings.warn(f"File {model_path} is not a JIT archive. Loading as a state dict instead")
 72 |             jit = False
 73 |         state_dict = torch.load(model_path, map_location="cpu")
 74 | 
 75 |     if not jit:
 76 |         # Build a non-jit model from the OpenAI jitted model state dict
 77 |         cast_dtype = get_cast_dtype(precision)
 78 |         try:
 79 |             model = build_model_from_openai_state_dict(state_dict or model.state_dict(), cast_dtype=cast_dtype)
 80 |         except KeyError:
 81 |             sd = {k[7:]: v for k, v in state_dict["state_dict"].items()}
 82 |             model = build_model_from_openai_state_dict(sd, cast_dtype=cast_dtype)
 83 | 
 84 |         # model from OpenAI state dict is in manually cast fp16 mode, must be converted for AMP/fp32/bf16 use
 85 |         model = model.to(device)
 86 |         if precision.startswith("amp") or precision == "fp32":
 87 |             model.float()
 88 |         elif precision == "bf16":
 89 |             convert_weights_to_lp(model, dtype=torch.bfloat16)
 90 | 
 91 |         return model
 92 | 
 93 |     # patch the device names
 94 |     device_holder = torch.jit.trace(lambda: torch.ones([]).to(torch.device(device)), example_inputs=[])
 95 |     device_node = [n for n in device_holder.graph.findAllNodes("prim::Constant") if "Device" in repr(n)][-1]
 96 | 
 97 |     def patch_device(module):
 98 |         try:
 99 |             graphs = [module.graph] if hasattr(module, "graph") else []
100 |         except RuntimeError:
101 |             graphs = []
102 | 
103 |         if hasattr(module, "forward1"):
104 |             graphs.append(module.forward1.graph)
105 | 
106 |         for graph in graphs:
107 |             for node in graph.findAllNodes("prim::Constant"):
108 |                 if "value" in node.attributeNames() and str(node["value"]).startswith("cuda"):
109 |                     node.copyAttributes(device_node)
110 | 
111 |     model.apply(patch_device)
112 |     patch_device(model.encode_image)
113 |     patch_device(model.encode_text)
114 | 
115 |     # patch dtype to float32 (typically for CPU)
116 |     if precision == "fp32":
117 |         float_holder = torch.jit.trace(lambda: torch.ones([]).float(), example_inputs=[])
118 |         float_input = list(float_holder.graph.findNode("aten::to").inputs())[1]
119 |         float_node = float_input.node()
120 | 
121 |         def patch_float(module):
122 |             try:
123 |                 graphs = [module.graph] if hasattr(module, "graph") else []
124 |             except RuntimeError:
125 |                 graphs = []
126 | 
127 |             if hasattr(module, "forward1"):
128 |                 graphs.append(module.forward1.graph)
129 | 
130 |             for graph in graphs:
131 |                 for node in graph.findAllNodes("aten::to"):
132 |                     inputs = list(node.inputs())
133 |                     for i in [1, 2]:  # dtype can be the second or third argument to aten::to()
134 |                         if inputs[i].node()["value"] == 5:
135 |                             inputs[i].node().copyAttributes(float_node)
136 | 
137 |         model.apply(patch_float)
138 |         patch_float(model.encode_image)
139 |         patch_float(model.encode_text)
140 |         model.float()
141 | 
142 |     # ensure image_size attr available at consistent location for both jit and non-jit
143 |     model.visual.image_size = model.input_resolution.item()
144 |     return model
145 | 


--------------------------------------------------------------------------------