Instructions to use prithivMLmods/ImageShield-SUPER-90M with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use prithivMLmods/ImageShield-SUPER-90M with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("image-classification", model="prithivMLmods/ImageShield-SUPER-90M") pipe("https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/hub/parrots.png")# Load model directly from transformers import AutoProcessor, AutoModelForImageClassification processor = AutoProcessor.from_pretrained("prithivMLmods/ImageShield-SUPER-90M") model = AutoModelForImageClassification.from_pretrained("prithivMLmods/ImageShield-SUPER-90M", device_map="auto") - Notebooks
- Google Colab
- Kaggle
Add files using upload-large-folder tool
Browse files- .gitattributes +1 -0
- assets/class_distribution_pie.png +0 -0
- assets/classification_report_bar.png +0 -0
- assets/confusion_matrix.png +0 -0
- assets/misalignment_distribution_pie.png +0 -0
- assets/prediction_accuracy_pie.png +0 -0
- assets/training_eval_graph.png +3 -0
- checkpoint-1500/config.json +50 -0
- checkpoint-1500/model.safetensors +3 -0
- checkpoint-1500/optimizer.pt +3 -0
- checkpoint-1500/preprocessor_config.json +23 -0
- checkpoint-1500/rng_state.pth +3 -0
- checkpoint-1500/scheduler.pt +3 -0
- checkpoint-1500/trainer_state.json +65 -0
- checkpoint-1500/training_args.bin +3 -0
- checkpoint-3000/config.json +50 -0
- checkpoint-3000/model.safetensors +3 -0
- checkpoint-3000/optimizer.pt +3 -0
- checkpoint-3000/preprocessor_config.json +23 -0
- checkpoint-3000/rng_state.pth +3 -0
- checkpoint-3000/scheduler.pt +3 -0
- checkpoint-3000/trainer_state.json +96 -0
- checkpoint-3000/training_args.bin +3 -0
- checkpoint-4500/config.json +50 -0
- checkpoint-4500/model.safetensors +3 -0
- checkpoint-4500/optimizer.pt +3 -0
- checkpoint-4500/preprocessor_config.json +23 -0
- checkpoint-4500/rng_state.pth +3 -0
- checkpoint-4500/scheduler.pt +3 -0
- checkpoint-4500/trainer_state.json +127 -0
- checkpoint-4500/training_args.bin +3 -0
- checkpoint-6000/config.json +50 -0
- checkpoint-6000/model.safetensors +3 -0
- checkpoint-6000/optimizer.pt +3 -0
- checkpoint-6000/preprocessor_config.json +23 -0
- checkpoint-6000/rng_state.pth +3 -0
- checkpoint-6000/scheduler.pt +3 -0
- checkpoint-6000/trainer_state.json +158 -0
- checkpoint-6000/training_args.bin +3 -0
- config.json +50 -0
- model.safetensors +3 -0
- preprocessor_config.json +23 -0
- results/report.txt +11 -0
- training_args.bin +3 -0
.gitattributes
CHANGED
|
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
assets/training_eval_graph.png filter=lfs diff=lfs merge=lfs -text
|
assets/class_distribution_pie.png
ADDED
|
assets/classification_report_bar.png
ADDED
|
assets/confusion_matrix.png
ADDED
|
assets/misalignment_distribution_pie.png
ADDED
|
assets/prediction_accuracy_pie.png
ADDED
|
assets/training_eval_graph.png
ADDED
|
Git LFS Details
|
checkpoint-1500/config.json
ADDED
|
@@ -0,0 +1,50 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"architectures": [
|
| 3 |
+
"SiglipForImageClassification"
|
| 4 |
+
],
|
| 5 |
+
"dtype": "float32",
|
| 6 |
+
"id2label": {
|
| 7 |
+
"0": "Safe",
|
| 8 |
+
"1": "Unsafe"
|
| 9 |
+
},
|
| 10 |
+
"initializer_factor": 1.0,
|
| 11 |
+
"label2id": {
|
| 12 |
+
"Safe": 0,
|
| 13 |
+
"Unsafe": 1
|
| 14 |
+
},
|
| 15 |
+
"model_type": "siglip",
|
| 16 |
+
"problem_type": "single_label_classification",
|
| 17 |
+
"text_config": {
|
| 18 |
+
"attention_dropout": 0.0,
|
| 19 |
+
"bos_token_id": 49406,
|
| 20 |
+
"dtype": "float32",
|
| 21 |
+
"eos_token_id": 49407,
|
| 22 |
+
"hidden_act": "gelu_pytorch_tanh",
|
| 23 |
+
"hidden_size": 768,
|
| 24 |
+
"intermediate_size": 3072,
|
| 25 |
+
"layer_norm_eps": 1e-06,
|
| 26 |
+
"max_position_embeddings": 64,
|
| 27 |
+
"model_type": "siglip_text_model",
|
| 28 |
+
"num_attention_heads": 12,
|
| 29 |
+
"num_hidden_layers": 12,
|
| 30 |
+
"pad_token_id": 1,
|
| 31 |
+
"projection_size": 768,
|
| 32 |
+
"vocab_size": 256000
|
| 33 |
+
},
|
| 34 |
+
"transformers_version": "5.15.0",
|
| 35 |
+
"use_cache": false,
|
| 36 |
+
"vision_config": {
|
| 37 |
+
"attention_dropout": 0.0,
|
| 38 |
+
"dtype": "float32",
|
| 39 |
+
"hidden_act": "gelu_pytorch_tanh",
|
| 40 |
+
"hidden_size": 768,
|
| 41 |
+
"image_size": 224,
|
| 42 |
+
"intermediate_size": 3072,
|
| 43 |
+
"layer_norm_eps": 1e-06,
|
| 44 |
+
"model_type": "siglip_vision_model",
|
| 45 |
+
"num_attention_heads": 12,
|
| 46 |
+
"num_channels": 3,
|
| 47 |
+
"num_hidden_layers": 12,
|
| 48 |
+
"patch_size": 16
|
| 49 |
+
}
|
| 50 |
+
}
|
checkpoint-1500/model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1a4bf26b53cbcff0b6f4a3bec770a10efabbaa2423c2cdac63b425073c5d90e9
|
| 3 |
+
size 371567992
|
checkpoint-1500/optimizer.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:596e83388c7003d528a9bb0536d31f7cd9e15b6a10522598513ab8b4941f323f
|
| 3 |
+
size 686558987
|
checkpoint-1500/preprocessor_config.json
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"do_convert_rgb": null,
|
| 3 |
+
"do_normalize": true,
|
| 4 |
+
"do_rescale": true,
|
| 5 |
+
"do_resize": true,
|
| 6 |
+
"image_mean": [
|
| 7 |
+
0.5,
|
| 8 |
+
0.5,
|
| 9 |
+
0.5
|
| 10 |
+
],
|
| 11 |
+
"image_processor_type": "SiglipImageProcessor",
|
| 12 |
+
"image_std": [
|
| 13 |
+
0.5,
|
| 14 |
+
0.5,
|
| 15 |
+
0.5
|
| 16 |
+
],
|
| 17 |
+
"resample": 2,
|
| 18 |
+
"rescale_factor": 0.00392156862745098,
|
| 19 |
+
"size": {
|
| 20 |
+
"height": 224,
|
| 21 |
+
"width": 224
|
| 22 |
+
}
|
| 23 |
+
}
|
checkpoint-1500/rng_state.pth
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1fec776a01b7ecf149a6b1fa67bccfb1d6b77988fc3a05eb7372052d21fffb59
|
| 3 |
+
size 14645
|
checkpoint-1500/scheduler.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:7f80ce5b2896740197905819b11f7b8511b8342b145922d0782dfccc2e7d6ffb
|
| 3 |
+
size 1465
|
checkpoint-1500/trainer_state.json
ADDED
|
@@ -0,0 +1,65 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"best_global_step": 1500,
|
| 3 |
+
"best_metric": 0.37709423899650574,
|
| 4 |
+
"best_model_checkpoint": "siglip2-image-classification/checkpoint-1500",
|
| 5 |
+
"epoch": 1.0,
|
| 6 |
+
"eval_steps": 500,
|
| 7 |
+
"global_step": 1500,
|
| 8 |
+
"is_hyper_param_search": false,
|
| 9 |
+
"is_local_process_zero": true,
|
| 10 |
+
"is_world_process_zero": true,
|
| 11 |
+
"log_history": [
|
| 12 |
+
{
|
| 13 |
+
"epoch": 0.3333333333333333,
|
| 14 |
+
"grad_norm": 3.479196786880493,
|
| 15 |
+
"learning_rate": 0.00018490756302521008,
|
| 16 |
+
"loss": 0.5047473754882813,
|
| 17 |
+
"step": 500
|
| 18 |
+
},
|
| 19 |
+
{
|
| 20 |
+
"epoch": 0.6666666666666666,
|
| 21 |
+
"grad_norm": 4.285048007965088,
|
| 22 |
+
"learning_rate": 0.00016810084033613447,
|
| 23 |
+
"loss": 0.42015097045898436,
|
| 24 |
+
"step": 1000
|
| 25 |
+
},
|
| 26 |
+
{
|
| 27 |
+
"epoch": 1.0,
|
| 28 |
+
"grad_norm": 2.655521869659424,
|
| 29 |
+
"learning_rate": 0.00015129411764705882,
|
| 30 |
+
"loss": 0.38898434448242186,
|
| 31 |
+
"step": 1500
|
| 32 |
+
},
|
| 33 |
+
{
|
| 34 |
+
"epoch": 1.0,
|
| 35 |
+
"eval_accuracy": 0.8312134754211069,
|
| 36 |
+
"eval_loss": 0.37709423899650574,
|
| 37 |
+
"eval_model_preparation_time": 0.0019,
|
| 38 |
+
"eval_runtime": 283.3202,
|
| 39 |
+
"eval_samples_per_second": 112.943,
|
| 40 |
+
"eval_steps_per_second": 14.118,
|
| 41 |
+
"step": 1500
|
| 42 |
+
}
|
| 43 |
+
],
|
| 44 |
+
"logging_steps": 500,
|
| 45 |
+
"max_steps": 6000,
|
| 46 |
+
"num_input_tokens_seen": 0,
|
| 47 |
+
"num_train_epochs": 4,
|
| 48 |
+
"save_steps": 500,
|
| 49 |
+
"stateful_callbacks": {
|
| 50 |
+
"TrainerControl": {
|
| 51 |
+
"args": {
|
| 52 |
+
"should_epoch_stop": false,
|
| 53 |
+
"should_evaluate": false,
|
| 54 |
+
"should_log": false,
|
| 55 |
+
"should_save": true,
|
| 56 |
+
"should_training_stop": false
|
| 57 |
+
},
|
| 58 |
+
"attributes": {}
|
| 59 |
+
}
|
| 60 |
+
},
|
| 61 |
+
"total_flos": 4.0200962884313334e+18,
|
| 62 |
+
"train_batch_size": 32,
|
| 63 |
+
"trial_name": null,
|
| 64 |
+
"trial_params": null
|
| 65 |
+
}
|
checkpoint-1500/training_args.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:fe9e466d18e2bd615281c80583abd8151d5c8bc84623dad56b5e15f0d4c899ee
|
| 3 |
+
size 5201
|
checkpoint-3000/config.json
ADDED
|
@@ -0,0 +1,50 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"architectures": [
|
| 3 |
+
"SiglipForImageClassification"
|
| 4 |
+
],
|
| 5 |
+
"dtype": "float32",
|
| 6 |
+
"id2label": {
|
| 7 |
+
"0": "Safe",
|
| 8 |
+
"1": "Unsafe"
|
| 9 |
+
},
|
| 10 |
+
"initializer_factor": 1.0,
|
| 11 |
+
"label2id": {
|
| 12 |
+
"Safe": 0,
|
| 13 |
+
"Unsafe": 1
|
| 14 |
+
},
|
| 15 |
+
"model_type": "siglip",
|
| 16 |
+
"problem_type": "single_label_classification",
|
| 17 |
+
"text_config": {
|
| 18 |
+
"attention_dropout": 0.0,
|
| 19 |
+
"bos_token_id": 49406,
|
| 20 |
+
"dtype": "float32",
|
| 21 |
+
"eos_token_id": 49407,
|
| 22 |
+
"hidden_act": "gelu_pytorch_tanh",
|
| 23 |
+
"hidden_size": 768,
|
| 24 |
+
"intermediate_size": 3072,
|
| 25 |
+
"layer_norm_eps": 1e-06,
|
| 26 |
+
"max_position_embeddings": 64,
|
| 27 |
+
"model_type": "siglip_text_model",
|
| 28 |
+
"num_attention_heads": 12,
|
| 29 |
+
"num_hidden_layers": 12,
|
| 30 |
+
"pad_token_id": 1,
|
| 31 |
+
"projection_size": 768,
|
| 32 |
+
"vocab_size": 256000
|
| 33 |
+
},
|
| 34 |
+
"transformers_version": "5.15.0",
|
| 35 |
+
"use_cache": false,
|
| 36 |
+
"vision_config": {
|
| 37 |
+
"attention_dropout": 0.0,
|
| 38 |
+
"dtype": "float32",
|
| 39 |
+
"hidden_act": "gelu_pytorch_tanh",
|
| 40 |
+
"hidden_size": 768,
|
| 41 |
+
"image_size": 224,
|
| 42 |
+
"intermediate_size": 3072,
|
| 43 |
+
"layer_norm_eps": 1e-06,
|
| 44 |
+
"model_type": "siglip_vision_model",
|
| 45 |
+
"num_attention_heads": 12,
|
| 46 |
+
"num_channels": 3,
|
| 47 |
+
"num_hidden_layers": 12,
|
| 48 |
+
"patch_size": 16
|
| 49 |
+
}
|
| 50 |
+
}
|
checkpoint-3000/model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:84c3d8ec730897768c694c4c0cdc5854cdb146532987a0dde4d6ec503a37565a
|
| 3 |
+
size 371567992
|
checkpoint-3000/optimizer.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ac81a435b16596ccca147f836058a98f63cb5fbdcf6ae54287d348b317d99e67
|
| 3 |
+
size 686558987
|
checkpoint-3000/preprocessor_config.json
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"do_convert_rgb": null,
|
| 3 |
+
"do_normalize": true,
|
| 4 |
+
"do_rescale": true,
|
| 5 |
+
"do_resize": true,
|
| 6 |
+
"image_mean": [
|
| 7 |
+
0.5,
|
| 8 |
+
0.5,
|
| 9 |
+
0.5
|
| 10 |
+
],
|
| 11 |
+
"image_processor_type": "SiglipImageProcessor",
|
| 12 |
+
"image_std": [
|
| 13 |
+
0.5,
|
| 14 |
+
0.5,
|
| 15 |
+
0.5
|
| 16 |
+
],
|
| 17 |
+
"resample": 2,
|
| 18 |
+
"rescale_factor": 0.00392156862745098,
|
| 19 |
+
"size": {
|
| 20 |
+
"height": 224,
|
| 21 |
+
"width": 224
|
| 22 |
+
}
|
| 23 |
+
}
|
checkpoint-3000/rng_state.pth
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:6fc2d9ff41fb99781f8df34958e6a2b2f9e812af596cecf3dbc71061de56ddd7
|
| 3 |
+
size 14645
|
checkpoint-3000/scheduler.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:22571f75ee2018c3049a09458bbe6c6c1231f3f666682ee6fdc5481ec5a4aacb
|
| 3 |
+
size 1465
|
checkpoint-3000/trainer_state.json
ADDED
|
@@ -0,0 +1,96 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"best_global_step": 3000,
|
| 3 |
+
"best_metric": 0.3550371527671814,
|
| 4 |
+
"best_model_checkpoint": "siglip2-image-classification/checkpoint-3000",
|
| 5 |
+
"epoch": 2.0,
|
| 6 |
+
"eval_steps": 500,
|
| 7 |
+
"global_step": 3000,
|
| 8 |
+
"is_hyper_param_search": false,
|
| 9 |
+
"is_local_process_zero": true,
|
| 10 |
+
"is_world_process_zero": true,
|
| 11 |
+
"log_history": [
|
| 12 |
+
{
|
| 13 |
+
"epoch": 0.3333333333333333,
|
| 14 |
+
"grad_norm": 3.479196786880493,
|
| 15 |
+
"learning_rate": 0.00018490756302521008,
|
| 16 |
+
"loss": 0.5047473754882813,
|
| 17 |
+
"step": 500
|
| 18 |
+
},
|
| 19 |
+
{
|
| 20 |
+
"epoch": 0.6666666666666666,
|
| 21 |
+
"grad_norm": 4.285048007965088,
|
| 22 |
+
"learning_rate": 0.00016810084033613447,
|
| 23 |
+
"loss": 0.42015097045898436,
|
| 24 |
+
"step": 1000
|
| 25 |
+
},
|
| 26 |
+
{
|
| 27 |
+
"epoch": 1.0,
|
| 28 |
+
"grad_norm": 2.655521869659424,
|
| 29 |
+
"learning_rate": 0.00015129411764705882,
|
| 30 |
+
"loss": 0.38898434448242186,
|
| 31 |
+
"step": 1500
|
| 32 |
+
},
|
| 33 |
+
{
|
| 34 |
+
"epoch": 1.0,
|
| 35 |
+
"eval_accuracy": 0.8312134754211069,
|
| 36 |
+
"eval_loss": 0.37709423899650574,
|
| 37 |
+
"eval_model_preparation_time": 0.0019,
|
| 38 |
+
"eval_runtime": 283.3202,
|
| 39 |
+
"eval_samples_per_second": 112.943,
|
| 40 |
+
"eval_steps_per_second": 14.118,
|
| 41 |
+
"step": 1500
|
| 42 |
+
},
|
| 43 |
+
{
|
| 44 |
+
"epoch": 1.3333333333333333,
|
| 45 |
+
"grad_norm": 1.3074887990951538,
|
| 46 |
+
"learning_rate": 0.0001344873949579832,
|
| 47 |
+
"loss": 0.3693521728515625,
|
| 48 |
+
"step": 2000
|
| 49 |
+
},
|
| 50 |
+
{
|
| 51 |
+
"epoch": 1.6666666666666665,
|
| 52 |
+
"grad_norm": 1.2879441976547241,
|
| 53 |
+
"learning_rate": 0.00011768067226890757,
|
| 54 |
+
"loss": 0.3484758605957031,
|
| 55 |
+
"step": 2500
|
| 56 |
+
},
|
| 57 |
+
{
|
| 58 |
+
"epoch": 2.0,
|
| 59 |
+
"grad_norm": 1.5275274515151978,
|
| 60 |
+
"learning_rate": 0.00010087394957983194,
|
| 61 |
+
"loss": 0.34061029052734376,
|
| 62 |
+
"step": 3000
|
| 63 |
+
},
|
| 64 |
+
{
|
| 65 |
+
"epoch": 2.0,
|
| 66 |
+
"eval_accuracy": 0.8569642801337541,
|
| 67 |
+
"eval_loss": 0.3550371527671814,
|
| 68 |
+
"eval_model_preparation_time": 0.0019,
|
| 69 |
+
"eval_runtime": 282.8579,
|
| 70 |
+
"eval_samples_per_second": 113.127,
|
| 71 |
+
"eval_steps_per_second": 14.141,
|
| 72 |
+
"step": 3000
|
| 73 |
+
}
|
| 74 |
+
],
|
| 75 |
+
"logging_steps": 500,
|
| 76 |
+
"max_steps": 6000,
|
| 77 |
+
"num_input_tokens_seen": 0,
|
| 78 |
+
"num_train_epochs": 4,
|
| 79 |
+
"save_steps": 500,
|
| 80 |
+
"stateful_callbacks": {
|
| 81 |
+
"TrainerControl": {
|
| 82 |
+
"args": {
|
| 83 |
+
"should_epoch_stop": false,
|
| 84 |
+
"should_evaluate": false,
|
| 85 |
+
"should_log": false,
|
| 86 |
+
"should_save": true,
|
| 87 |
+
"should_training_stop": false
|
| 88 |
+
},
|
| 89 |
+
"attributes": {}
|
| 90 |
+
}
|
| 91 |
+
},
|
| 92 |
+
"total_flos": 8.040192576862667e+18,
|
| 93 |
+
"train_batch_size": 32,
|
| 94 |
+
"trial_name": null,
|
| 95 |
+
"trial_params": null
|
| 96 |
+
}
|
checkpoint-3000/training_args.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:fe9e466d18e2bd615281c80583abd8151d5c8bc84623dad56b5e15f0d4c899ee
|
| 3 |
+
size 5201
|
checkpoint-4500/config.json
ADDED
|
@@ -0,0 +1,50 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"architectures": [
|
| 3 |
+
"SiglipForImageClassification"
|
| 4 |
+
],
|
| 5 |
+
"dtype": "float32",
|
| 6 |
+
"id2label": {
|
| 7 |
+
"0": "Safe",
|
| 8 |
+
"1": "Unsafe"
|
| 9 |
+
},
|
| 10 |
+
"initializer_factor": 1.0,
|
| 11 |
+
"label2id": {
|
| 12 |
+
"Safe": 0,
|
| 13 |
+
"Unsafe": 1
|
| 14 |
+
},
|
| 15 |
+
"model_type": "siglip",
|
| 16 |
+
"problem_type": "single_label_classification",
|
| 17 |
+
"text_config": {
|
| 18 |
+
"attention_dropout": 0.0,
|
| 19 |
+
"bos_token_id": 49406,
|
| 20 |
+
"dtype": "float32",
|
| 21 |
+
"eos_token_id": 49407,
|
| 22 |
+
"hidden_act": "gelu_pytorch_tanh",
|
| 23 |
+
"hidden_size": 768,
|
| 24 |
+
"intermediate_size": 3072,
|
| 25 |
+
"layer_norm_eps": 1e-06,
|
| 26 |
+
"max_position_embeddings": 64,
|
| 27 |
+
"model_type": "siglip_text_model",
|
| 28 |
+
"num_attention_heads": 12,
|
| 29 |
+
"num_hidden_layers": 12,
|
| 30 |
+
"pad_token_id": 1,
|
| 31 |
+
"projection_size": 768,
|
| 32 |
+
"vocab_size": 256000
|
| 33 |
+
},
|
| 34 |
+
"transformers_version": "5.15.0",
|
| 35 |
+
"use_cache": false,
|
| 36 |
+
"vision_config": {
|
| 37 |
+
"attention_dropout": 0.0,
|
| 38 |
+
"dtype": "float32",
|
| 39 |
+
"hidden_act": "gelu_pytorch_tanh",
|
| 40 |
+
"hidden_size": 768,
|
| 41 |
+
"image_size": 224,
|
| 42 |
+
"intermediate_size": 3072,
|
| 43 |
+
"layer_norm_eps": 1e-06,
|
| 44 |
+
"model_type": "siglip_vision_model",
|
| 45 |
+
"num_attention_heads": 12,
|
| 46 |
+
"num_channels": 3,
|
| 47 |
+
"num_hidden_layers": 12,
|
| 48 |
+
"patch_size": 16
|
| 49 |
+
}
|
| 50 |
+
}
|
checkpoint-4500/model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f08a49c02e63f8ed0272e48f0425cfa98e203356333b19cdf52b77265f081278
|
| 3 |
+
size 371567992
|
checkpoint-4500/optimizer.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:baf9fd75acfa623d1a667b661965dd586163ebd551a0b6081177457eb1b75307
|
| 3 |
+
size 686558987
|
checkpoint-4500/preprocessor_config.json
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"do_convert_rgb": null,
|
| 3 |
+
"do_normalize": true,
|
| 4 |
+
"do_rescale": true,
|
| 5 |
+
"do_resize": true,
|
| 6 |
+
"image_mean": [
|
| 7 |
+
0.5,
|
| 8 |
+
0.5,
|
| 9 |
+
0.5
|
| 10 |
+
],
|
| 11 |
+
"image_processor_type": "SiglipImageProcessor",
|
| 12 |
+
"image_std": [
|
| 13 |
+
0.5,
|
| 14 |
+
0.5,
|
| 15 |
+
0.5
|
| 16 |
+
],
|
| 17 |
+
"resample": 2,
|
| 18 |
+
"rescale_factor": 0.00392156862745098,
|
| 19 |
+
"size": {
|
| 20 |
+
"height": 224,
|
| 21 |
+
"width": 224
|
| 22 |
+
}
|
| 23 |
+
}
|
checkpoint-4500/rng_state.pth
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:8a229ed0ae2d399ab1847c127a24a2743b2b12149859e0cb5c90c1d8fa8ece68
|
| 3 |
+
size 14645
|
checkpoint-4500/scheduler.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:fda5005268f2e2d020512b396c9904946a7fa4a8525e5969867d90876d389218
|
| 3 |
+
size 1465
|
checkpoint-4500/trainer_state.json
ADDED
|
@@ -0,0 +1,127 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"best_global_step": 4500,
|
| 3 |
+
"best_metric": 0.2799703776836395,
|
| 4 |
+
"best_model_checkpoint": "siglip2-image-classification/checkpoint-4500",
|
| 5 |
+
"epoch": 3.0,
|
| 6 |
+
"eval_steps": 500,
|
| 7 |
+
"global_step": 4500,
|
| 8 |
+
"is_hyper_param_search": false,
|
| 9 |
+
"is_local_process_zero": true,
|
| 10 |
+
"is_world_process_zero": true,
|
| 11 |
+
"log_history": [
|
| 12 |
+
{
|
| 13 |
+
"epoch": 0.3333333333333333,
|
| 14 |
+
"grad_norm": 3.479196786880493,
|
| 15 |
+
"learning_rate": 0.00018490756302521008,
|
| 16 |
+
"loss": 0.5047473754882813,
|
| 17 |
+
"step": 500
|
| 18 |
+
},
|
| 19 |
+
{
|
| 20 |
+
"epoch": 0.6666666666666666,
|
| 21 |
+
"grad_norm": 4.285048007965088,
|
| 22 |
+
"learning_rate": 0.00016810084033613447,
|
| 23 |
+
"loss": 0.42015097045898436,
|
| 24 |
+
"step": 1000
|
| 25 |
+
},
|
| 26 |
+
{
|
| 27 |
+
"epoch": 1.0,
|
| 28 |
+
"grad_norm": 2.655521869659424,
|
| 29 |
+
"learning_rate": 0.00015129411764705882,
|
| 30 |
+
"loss": 0.38898434448242186,
|
| 31 |
+
"step": 1500
|
| 32 |
+
},
|
| 33 |
+
{
|
| 34 |
+
"epoch": 1.0,
|
| 35 |
+
"eval_accuracy": 0.8312134754211069,
|
| 36 |
+
"eval_loss": 0.37709423899650574,
|
| 37 |
+
"eval_model_preparation_time": 0.0019,
|
| 38 |
+
"eval_runtime": 283.3202,
|
| 39 |
+
"eval_samples_per_second": 112.943,
|
| 40 |
+
"eval_steps_per_second": 14.118,
|
| 41 |
+
"step": 1500
|
| 42 |
+
},
|
| 43 |
+
{
|
| 44 |
+
"epoch": 1.3333333333333333,
|
| 45 |
+
"grad_norm": 1.3074887990951538,
|
| 46 |
+
"learning_rate": 0.0001344873949579832,
|
| 47 |
+
"loss": 0.3693521728515625,
|
| 48 |
+
"step": 2000
|
| 49 |
+
},
|
| 50 |
+
{
|
| 51 |
+
"epoch": 1.6666666666666665,
|
| 52 |
+
"grad_norm": 1.2879441976547241,
|
| 53 |
+
"learning_rate": 0.00011768067226890757,
|
| 54 |
+
"loss": 0.3484758605957031,
|
| 55 |
+
"step": 2500
|
| 56 |
+
},
|
| 57 |
+
{
|
| 58 |
+
"epoch": 2.0,
|
| 59 |
+
"grad_norm": 1.5275274515151978,
|
| 60 |
+
"learning_rate": 0.00010087394957983194,
|
| 61 |
+
"loss": 0.34061029052734376,
|
| 62 |
+
"step": 3000
|
| 63 |
+
},
|
| 64 |
+
{
|
| 65 |
+
"epoch": 2.0,
|
| 66 |
+
"eval_accuracy": 0.8569642801337541,
|
| 67 |
+
"eval_loss": 0.3550371527671814,
|
| 68 |
+
"eval_model_preparation_time": 0.0019,
|
| 69 |
+
"eval_runtime": 282.8579,
|
| 70 |
+
"eval_samples_per_second": 113.127,
|
| 71 |
+
"eval_steps_per_second": 14.141,
|
| 72 |
+
"step": 3000
|
| 73 |
+
},
|
| 74 |
+
{
|
| 75 |
+
"epoch": 2.3333333333333335,
|
| 76 |
+
"grad_norm": 4.5269551277160645,
|
| 77 |
+
"learning_rate": 8.406722689075631e-05,
|
| 78 |
+
"loss": 0.31870977783203125,
|
| 79 |
+
"step": 3500
|
| 80 |
+
},
|
| 81 |
+
{
|
| 82 |
+
"epoch": 2.6666666666666665,
|
| 83 |
+
"grad_norm": 1.9202643632888794,
|
| 84 |
+
"learning_rate": 6.726050420168067e-05,
|
| 85 |
+
"loss": 0.30601947021484377,
|
| 86 |
+
"step": 4000
|
| 87 |
+
},
|
| 88 |
+
{
|
| 89 |
+
"epoch": 3.0,
|
| 90 |
+
"grad_norm": 2.203441858291626,
|
| 91 |
+
"learning_rate": 5.045378151260505e-05,
|
| 92 |
+
"loss": 0.2938027648925781,
|
| 93 |
+
"step": 4500
|
| 94 |
+
},
|
| 95 |
+
{
|
| 96 |
+
"epoch": 3.0,
|
| 97 |
+
"eval_accuracy": 0.8811525360167505,
|
| 98 |
+
"eval_loss": 0.2799703776836395,
|
| 99 |
+
"eval_model_preparation_time": 0.0019,
|
| 100 |
+
"eval_runtime": 287.094,
|
| 101 |
+
"eval_samples_per_second": 111.458,
|
| 102 |
+
"eval_steps_per_second": 13.933,
|
| 103 |
+
"step": 4500
|
| 104 |
+
}
|
| 105 |
+
],
|
| 106 |
+
"logging_steps": 500,
|
| 107 |
+
"max_steps": 6000,
|
| 108 |
+
"num_input_tokens_seen": 0,
|
| 109 |
+
"num_train_epochs": 4,
|
| 110 |
+
"save_steps": 500,
|
| 111 |
+
"stateful_callbacks": {
|
| 112 |
+
"TrainerControl": {
|
| 113 |
+
"args": {
|
| 114 |
+
"should_epoch_stop": false,
|
| 115 |
+
"should_evaluate": false,
|
| 116 |
+
"should_log": false,
|
| 117 |
+
"should_save": true,
|
| 118 |
+
"should_training_stop": false
|
| 119 |
+
},
|
| 120 |
+
"attributes": {}
|
| 121 |
+
}
|
| 122 |
+
},
|
| 123 |
+
"total_flos": 1.2060288865294e+19,
|
| 124 |
+
"train_batch_size": 32,
|
| 125 |
+
"trial_name": null,
|
| 126 |
+
"trial_params": null
|
| 127 |
+
}
|
checkpoint-4500/training_args.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:fe9e466d18e2bd615281c80583abd8151d5c8bc84623dad56b5e15f0d4c899ee
|
| 3 |
+
size 5201
|
checkpoint-6000/config.json
ADDED
|
@@ -0,0 +1,50 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"architectures": [
|
| 3 |
+
"SiglipForImageClassification"
|
| 4 |
+
],
|
| 5 |
+
"dtype": "float32",
|
| 6 |
+
"id2label": {
|
| 7 |
+
"0": "Safe",
|
| 8 |
+
"1": "Unsafe"
|
| 9 |
+
},
|
| 10 |
+
"initializer_factor": 1.0,
|
| 11 |
+
"label2id": {
|
| 12 |
+
"Safe": 0,
|
| 13 |
+
"Unsafe": 1
|
| 14 |
+
},
|
| 15 |
+
"model_type": "siglip",
|
| 16 |
+
"problem_type": "single_label_classification",
|
| 17 |
+
"text_config": {
|
| 18 |
+
"attention_dropout": 0.0,
|
| 19 |
+
"bos_token_id": 49406,
|
| 20 |
+
"dtype": "float32",
|
| 21 |
+
"eos_token_id": 49407,
|
| 22 |
+
"hidden_act": "gelu_pytorch_tanh",
|
| 23 |
+
"hidden_size": 768,
|
| 24 |
+
"intermediate_size": 3072,
|
| 25 |
+
"layer_norm_eps": 1e-06,
|
| 26 |
+
"max_position_embeddings": 64,
|
| 27 |
+
"model_type": "siglip_text_model",
|
| 28 |
+
"num_attention_heads": 12,
|
| 29 |
+
"num_hidden_layers": 12,
|
| 30 |
+
"pad_token_id": 1,
|
| 31 |
+
"projection_size": 768,
|
| 32 |
+
"vocab_size": 256000
|
| 33 |
+
},
|
| 34 |
+
"transformers_version": "5.15.0",
|
| 35 |
+
"use_cache": false,
|
| 36 |
+
"vision_config": {
|
| 37 |
+
"attention_dropout": 0.0,
|
| 38 |
+
"dtype": "float32",
|
| 39 |
+
"hidden_act": "gelu_pytorch_tanh",
|
| 40 |
+
"hidden_size": 768,
|
| 41 |
+
"image_size": 224,
|
| 42 |
+
"intermediate_size": 3072,
|
| 43 |
+
"layer_norm_eps": 1e-06,
|
| 44 |
+
"model_type": "siglip_vision_model",
|
| 45 |
+
"num_attention_heads": 12,
|
| 46 |
+
"num_channels": 3,
|
| 47 |
+
"num_hidden_layers": 12,
|
| 48 |
+
"patch_size": 16
|
| 49 |
+
}
|
| 50 |
+
}
|
checkpoint-6000/model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:25339db142dbeb2ed734696f6d099da305318d044f83f45145fc778c2f59bc17
|
| 3 |
+
size 371567992
|
checkpoint-6000/optimizer.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9d9f44539117d3865deea4a332d2112b86f25ac942a8e27c501e43447ad02f53
|
| 3 |
+
size 686558987
|
checkpoint-6000/preprocessor_config.json
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"do_convert_rgb": null,
|
| 3 |
+
"do_normalize": true,
|
| 4 |
+
"do_rescale": true,
|
| 5 |
+
"do_resize": true,
|
| 6 |
+
"image_mean": [
|
| 7 |
+
0.5,
|
| 8 |
+
0.5,
|
| 9 |
+
0.5
|
| 10 |
+
],
|
| 11 |
+
"image_processor_type": "SiglipImageProcessor",
|
| 12 |
+
"image_std": [
|
| 13 |
+
0.5,
|
| 14 |
+
0.5,
|
| 15 |
+
0.5
|
| 16 |
+
],
|
| 17 |
+
"resample": 2,
|
| 18 |
+
"rescale_factor": 0.00392156862745098,
|
| 19 |
+
"size": {
|
| 20 |
+
"height": 224,
|
| 21 |
+
"width": 224
|
| 22 |
+
}
|
| 23 |
+
}
|
checkpoint-6000/rng_state.pth
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:6325ce64fa1116d0b2384d58a7d2de831adbfb9513edd5786abb3ac15c7867f2
|
| 3 |
+
size 14645
|
checkpoint-6000/scheduler.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1f0326c04dcdb07f5852da1801a6efbb41ba882da2637d7d10449bb43a8b585a
|
| 3 |
+
size 1465
|
checkpoint-6000/trainer_state.json
ADDED
|
@@ -0,0 +1,158 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"best_global_step": 6000,
|
| 3 |
+
"best_metric": 0.25916826725006104,
|
| 4 |
+
"best_model_checkpoint": "siglip2-image-classification/checkpoint-6000",
|
| 5 |
+
"epoch": 4.0,
|
| 6 |
+
"eval_steps": 500,
|
| 7 |
+
"global_step": 6000,
|
| 8 |
+
"is_hyper_param_search": false,
|
| 9 |
+
"is_local_process_zero": true,
|
| 10 |
+
"is_world_process_zero": true,
|
| 11 |
+
"log_history": [
|
| 12 |
+
{
|
| 13 |
+
"epoch": 0.3333333333333333,
|
| 14 |
+
"grad_norm": 3.479196786880493,
|
| 15 |
+
"learning_rate": 0.00018490756302521008,
|
| 16 |
+
"loss": 0.5047473754882813,
|
| 17 |
+
"step": 500
|
| 18 |
+
},
|
| 19 |
+
{
|
| 20 |
+
"epoch": 0.6666666666666666,
|
| 21 |
+
"grad_norm": 4.285048007965088,
|
| 22 |
+
"learning_rate": 0.00016810084033613447,
|
| 23 |
+
"loss": 0.42015097045898436,
|
| 24 |
+
"step": 1000
|
| 25 |
+
},
|
| 26 |
+
{
|
| 27 |
+
"epoch": 1.0,
|
| 28 |
+
"grad_norm": 2.655521869659424,
|
| 29 |
+
"learning_rate": 0.00015129411764705882,
|
| 30 |
+
"loss": 0.38898434448242186,
|
| 31 |
+
"step": 1500
|
| 32 |
+
},
|
| 33 |
+
{
|
| 34 |
+
"epoch": 1.0,
|
| 35 |
+
"eval_accuracy": 0.8312134754211069,
|
| 36 |
+
"eval_loss": 0.37709423899650574,
|
| 37 |
+
"eval_model_preparation_time": 0.0019,
|
| 38 |
+
"eval_runtime": 283.3202,
|
| 39 |
+
"eval_samples_per_second": 112.943,
|
| 40 |
+
"eval_steps_per_second": 14.118,
|
| 41 |
+
"step": 1500
|
| 42 |
+
},
|
| 43 |
+
{
|
| 44 |
+
"epoch": 1.3333333333333333,
|
| 45 |
+
"grad_norm": 1.3074887990951538,
|
| 46 |
+
"learning_rate": 0.0001344873949579832,
|
| 47 |
+
"loss": 0.3693521728515625,
|
| 48 |
+
"step": 2000
|
| 49 |
+
},
|
| 50 |
+
{
|
| 51 |
+
"epoch": 1.6666666666666665,
|
| 52 |
+
"grad_norm": 1.2879441976547241,
|
| 53 |
+
"learning_rate": 0.00011768067226890757,
|
| 54 |
+
"loss": 0.3484758605957031,
|
| 55 |
+
"step": 2500
|
| 56 |
+
},
|
| 57 |
+
{
|
| 58 |
+
"epoch": 2.0,
|
| 59 |
+
"grad_norm": 1.5275274515151978,
|
| 60 |
+
"learning_rate": 0.00010087394957983194,
|
| 61 |
+
"loss": 0.34061029052734376,
|
| 62 |
+
"step": 3000
|
| 63 |
+
},
|
| 64 |
+
{
|
| 65 |
+
"epoch": 2.0,
|
| 66 |
+
"eval_accuracy": 0.8569642801337541,
|
| 67 |
+
"eval_loss": 0.3550371527671814,
|
| 68 |
+
"eval_model_preparation_time": 0.0019,
|
| 69 |
+
"eval_runtime": 282.8579,
|
| 70 |
+
"eval_samples_per_second": 113.127,
|
| 71 |
+
"eval_steps_per_second": 14.141,
|
| 72 |
+
"step": 3000
|
| 73 |
+
},
|
| 74 |
+
{
|
| 75 |
+
"epoch": 2.3333333333333335,
|
| 76 |
+
"grad_norm": 4.5269551277160645,
|
| 77 |
+
"learning_rate": 8.406722689075631e-05,
|
| 78 |
+
"loss": 0.31870977783203125,
|
| 79 |
+
"step": 3500
|
| 80 |
+
},
|
| 81 |
+
{
|
| 82 |
+
"epoch": 2.6666666666666665,
|
| 83 |
+
"grad_norm": 1.9202643632888794,
|
| 84 |
+
"learning_rate": 6.726050420168067e-05,
|
| 85 |
+
"loss": 0.30601947021484377,
|
| 86 |
+
"step": 4000
|
| 87 |
+
},
|
| 88 |
+
{
|
| 89 |
+
"epoch": 3.0,
|
| 90 |
+
"grad_norm": 2.203441858291626,
|
| 91 |
+
"learning_rate": 5.045378151260505e-05,
|
| 92 |
+
"loss": 0.2938027648925781,
|
| 93 |
+
"step": 4500
|
| 94 |
+
},
|
| 95 |
+
{
|
| 96 |
+
"epoch": 3.0,
|
| 97 |
+
"eval_accuracy": 0.8811525360167505,
|
| 98 |
+
"eval_loss": 0.2799703776836395,
|
| 99 |
+
"eval_model_preparation_time": 0.0019,
|
| 100 |
+
"eval_runtime": 287.094,
|
| 101 |
+
"eval_samples_per_second": 111.458,
|
| 102 |
+
"eval_steps_per_second": 13.933,
|
| 103 |
+
"step": 4500
|
| 104 |
+
},
|
| 105 |
+
{
|
| 106 |
+
"epoch": 3.3333333333333335,
|
| 107 |
+
"grad_norm": 3.2569119930267334,
|
| 108 |
+
"learning_rate": 3.364705882352941e-05,
|
| 109 |
+
"loss": 0.2793130798339844,
|
| 110 |
+
"step": 5000
|
| 111 |
+
},
|
| 112 |
+
{
|
| 113 |
+
"epoch": 3.6666666666666665,
|
| 114 |
+
"grad_norm": 3.2401018142700195,
|
| 115 |
+
"learning_rate": 1.6840336134453782e-05,
|
| 116 |
+
"loss": 0.26365960693359375,
|
| 117 |
+
"step": 5500
|
| 118 |
+
},
|
| 119 |
+
{
|
| 120 |
+
"epoch": 4.0,
|
| 121 |
+
"grad_norm": 1.6921696662902832,
|
| 122 |
+
"learning_rate": 3.3613445378151266e-08,
|
| 123 |
+
"loss": 0.2508096923828125,
|
| 124 |
+
"step": 6000
|
| 125 |
+
},
|
| 126 |
+
{
|
| 127 |
+
"epoch": 4.0,
|
| 128 |
+
"eval_accuracy": 0.8935591737241789,
|
| 129 |
+
"eval_loss": 0.25916826725006104,
|
| 130 |
+
"eval_model_preparation_time": 0.0019,
|
| 131 |
+
"eval_runtime": 286.6404,
|
| 132 |
+
"eval_samples_per_second": 111.635,
|
| 133 |
+
"eval_steps_per_second": 13.955,
|
| 134 |
+
"step": 6000
|
| 135 |
+
}
|
| 136 |
+
],
|
| 137 |
+
"logging_steps": 500,
|
| 138 |
+
"max_steps": 6000,
|
| 139 |
+
"num_input_tokens_seen": 0,
|
| 140 |
+
"num_train_epochs": 4,
|
| 141 |
+
"save_steps": 500,
|
| 142 |
+
"stateful_callbacks": {
|
| 143 |
+
"TrainerControl": {
|
| 144 |
+
"args": {
|
| 145 |
+
"should_epoch_stop": false,
|
| 146 |
+
"should_evaluate": false,
|
| 147 |
+
"should_log": false,
|
| 148 |
+
"should_save": true,
|
| 149 |
+
"should_training_stop": true
|
| 150 |
+
},
|
| 151 |
+
"attributes": {}
|
| 152 |
+
}
|
| 153 |
+
},
|
| 154 |
+
"total_flos": 1.6080385153725334e+19,
|
| 155 |
+
"train_batch_size": 32,
|
| 156 |
+
"trial_name": null,
|
| 157 |
+
"trial_params": null
|
| 158 |
+
}
|
checkpoint-6000/training_args.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:fe9e466d18e2bd615281c80583abd8151d5c8bc84623dad56b5e15f0d4c899ee
|
| 3 |
+
size 5201
|
config.json
ADDED
|
@@ -0,0 +1,50 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"architectures": [
|
| 3 |
+
"SiglipForImageClassification"
|
| 4 |
+
],
|
| 5 |
+
"dtype": "float32",
|
| 6 |
+
"id2label": {
|
| 7 |
+
"0": "Safe",
|
| 8 |
+
"1": "Unsafe"
|
| 9 |
+
},
|
| 10 |
+
"initializer_factor": 1.0,
|
| 11 |
+
"label2id": {
|
| 12 |
+
"Safe": 0,
|
| 13 |
+
"Unsafe": 1
|
| 14 |
+
},
|
| 15 |
+
"model_type": "siglip",
|
| 16 |
+
"problem_type": "single_label_classification",
|
| 17 |
+
"text_config": {
|
| 18 |
+
"attention_dropout": 0.0,
|
| 19 |
+
"bos_token_id": 49406,
|
| 20 |
+
"dtype": "float32",
|
| 21 |
+
"eos_token_id": 49407,
|
| 22 |
+
"hidden_act": "gelu_pytorch_tanh",
|
| 23 |
+
"hidden_size": 768,
|
| 24 |
+
"intermediate_size": 3072,
|
| 25 |
+
"layer_norm_eps": 1e-06,
|
| 26 |
+
"max_position_embeddings": 64,
|
| 27 |
+
"model_type": "siglip_text_model",
|
| 28 |
+
"num_attention_heads": 12,
|
| 29 |
+
"num_hidden_layers": 12,
|
| 30 |
+
"pad_token_id": 1,
|
| 31 |
+
"projection_size": 768,
|
| 32 |
+
"vocab_size": 256000
|
| 33 |
+
},
|
| 34 |
+
"transformers_version": "5.15.0",
|
| 35 |
+
"use_cache": false,
|
| 36 |
+
"vision_config": {
|
| 37 |
+
"attention_dropout": 0.0,
|
| 38 |
+
"dtype": "float32",
|
| 39 |
+
"hidden_act": "gelu_pytorch_tanh",
|
| 40 |
+
"hidden_size": 768,
|
| 41 |
+
"image_size": 224,
|
| 42 |
+
"intermediate_size": 3072,
|
| 43 |
+
"layer_norm_eps": 1e-06,
|
| 44 |
+
"model_type": "siglip_vision_model",
|
| 45 |
+
"num_attention_heads": 12,
|
| 46 |
+
"num_channels": 3,
|
| 47 |
+
"num_hidden_layers": 12,
|
| 48 |
+
"patch_size": 16
|
| 49 |
+
}
|
| 50 |
+
}
|
model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:25339db142dbeb2ed734696f6d099da305318d044f83f45145fc778c2f59bc17
|
| 3 |
+
size 371567992
|
preprocessor_config.json
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"do_convert_rgb": null,
|
| 3 |
+
"do_normalize": true,
|
| 4 |
+
"do_rescale": true,
|
| 5 |
+
"do_resize": true,
|
| 6 |
+
"image_mean": [
|
| 7 |
+
0.5,
|
| 8 |
+
0.5,
|
| 9 |
+
0.5
|
| 10 |
+
],
|
| 11 |
+
"image_processor_type": "SiglipImageProcessor",
|
| 12 |
+
"image_std": [
|
| 13 |
+
0.5,
|
| 14 |
+
0.5,
|
| 15 |
+
0.5
|
| 16 |
+
],
|
| 17 |
+
"resample": 2,
|
| 18 |
+
"rescale_factor": 0.00392156862745098,
|
| 19 |
+
"size": {
|
| 20 |
+
"height": 224,
|
| 21 |
+
"width": 224
|
| 22 |
+
}
|
| 23 |
+
}
|
results/report.txt
ADDED
|
@@ -0,0 +1,11 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Classification Report
|
| 2 |
+
============================================================
|
| 3 |
+
|
| 4 |
+
precision recall f1-score support
|
| 5 |
+
|
| 6 |
+
Safe 0.9128 0.8702 0.8910 16000
|
| 7 |
+
Unsafe 0.8760 0.9169 0.8960 15999
|
| 8 |
+
|
| 9 |
+
accuracy 0.8936 31999
|
| 10 |
+
macro avg 0.8944 0.8936 0.8935 31999
|
| 11 |
+
weighted avg 0.8944 0.8936 0.8935 31999
|
training_args.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:fe9e466d18e2bd615281c80583abd8151d5c8bc84623dad56b5e15f0d4c899ee
|
| 3 |
+
size 5201
|