brendaogutu commited on 4 days ago

Commit

bf318a5

verified ·

1 Parent(s): 028d3b5

Upload fine-tuned model - BLEU: 0.1720

Browse files

This view is limited to 50 files because it contains too many changes. See raw diff

Files changed (50) hide show

.gitattributes +8 -0
README.md +120 -0
baseline_results.json +15 -0
checkpoint-152/config.json +54 -0
checkpoint-152/generation_config.json +17 -0
checkpoint-152/model.safetensors +3 -0
checkpoint-152/optimizer.pt +3 -0
checkpoint-152/rng_state.pth +3 -0
checkpoint-152/scaler.pt +3 -0
checkpoint-152/scheduler.pt +3 -0
checkpoint-152/source.spm +3 -0
checkpoint-152/special_tokens_map.json +5 -0
checkpoint-152/target.spm +3 -0
checkpoint-152/tokenizer_config.json +39 -0
checkpoint-152/trainer_state.json +64 -0
checkpoint-152/training_args.bin +3 -0
checkpoint-152/vocab.json +0 -0
checkpoint-168/config.json +54 -0
checkpoint-168/generation_config.json +17 -0
checkpoint-168/model.safetensors +3 -0
checkpoint-168/optimizer.pt +3 -0
checkpoint-168/rng_state.pth +3 -0
checkpoint-168/scaler.pt +3 -0
checkpoint-168/scheduler.pt +3 -0
checkpoint-168/source.spm +3 -0
checkpoint-168/special_tokens_map.json +5 -0
checkpoint-168/target.spm +3 -0
checkpoint-168/tokenizer_config.json +39 -0
checkpoint-168/trainer_state.json +64 -0
checkpoint-168/training_args.bin +3 -0
checkpoint-168/vocab.json +0 -0
checkpoint-192/config.json +54 -0
checkpoint-192/generation_config.json +17 -0
checkpoint-192/model.safetensors +3 -0
checkpoint-192/optimizer.pt +3 -0
checkpoint-192/rng_state.pth +3 -0
checkpoint-192/scaler.pt +3 -0
checkpoint-192/scheduler.pt +3 -0
checkpoint-192/source.spm +3 -0
checkpoint-192/special_tokens_map.json +5 -0
checkpoint-192/target.spm +3 -0
checkpoint-192/tokenizer_config.json +39 -0
checkpoint-192/trainer_state.json +64 -0
checkpoint-192/training_args.bin +3 -0
checkpoint-192/vocab.json +0 -0
config.json +54 -0
generation_config.json +17 -0
model.safetensors +3 -0
source.spm +3 -0
special_tokens_map.json +5 -0

.gitattributes CHANGED Viewed

@@ -33,3 +33,11 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text

 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text
+checkpoint-152/source.spm filter=lfs diff=lfs merge=lfs -text
+checkpoint-152/target.spm filter=lfs diff=lfs merge=lfs -text
+checkpoint-168/source.spm filter=lfs diff=lfs merge=lfs -text
+checkpoint-168/target.spm filter=lfs diff=lfs merge=lfs -text
+checkpoint-192/source.spm filter=lfs diff=lfs merge=lfs -text
+checkpoint-192/target.spm filter=lfs diff=lfs merge=lfs -text
+source.spm filter=lfs diff=lfs merge=lfs -text
+target.spm filter=lfs diff=lfs merge=lfs -text

README.md ADDED Viewed

	@@ -0,0 +1,120 @@

+---
+language:
+- sw
+- en
+license: apache-2.0
+tags:
+- translation
+- swahili
+- english
+- opus-mt
+- openchs
+datasets:
+- nllb
+- ccaligned
+metrics:
+- bleu
+- chrf
+- comet
+---
+# Swahili-English Translation Model for Child Helpline Services
+## Model Description
+This model is a fine-tuned version of `Helsinki-NLP/opus-mt-mul-en` for Swahili-to-English translation, specifically optimized for child helpline call transcriptions in East Africa (Tanzania, Uganda, Kenya).
+**Developed by:** BITZ IT Consulting Ltd
+**Project:** OpenCHS (Open Child Helpline System)
+**Funded by:** UNICEF Venture Fund
+**License:** Apache 2.0
+## Training Data
+The model was fine-tuned on a combination of:
+- NLLB Swahili-English parallel corpus (high quality, weight: 1.0)
+- CCAligned web-crawled parallel data (supplementary, weight: 0.7)
+- Total training samples: Approximately 50,000+ sentence pairs
+- Domain focus: Conversational Swahili from helpline contexts
+## Performance
+### Test Set (General Translation)
+- **BLEU:** 0.1720
+- **chrF:** 36.54
+- **Improvement over baseline:** +0.0%
+### Domain Evaluation (Call Transcriptions)
+- **Domain BLEU:** 0.0000
+- **Domain chrF:** 1.88
+- **Domain COMET-QE:** 0.0000
+*Domain metrics are evaluated on real 10-minute call transcriptions from child helplines.*
+## Intended Use
+**Primary Use Case:** Translating Swahili helpline call transcriptions to English for:
+- Case documentation and reporting
+- Quality assurance and supervision
+- Cross-border case referrals
+- Data analysis and insights
+**Languages:** Swahili (source) → English (target)
+**Limitations:**
+- Optimized for conversational Swahili (East African dialects)
+- May not perform well on formal/literary Swahili
+- Best for text lengths under 512 tokens
+- Requires post-editing for critical use cases
+## Usage
+```python
+from transformers import MarianTokenizer, MarianMTModel
+model_name = "YOUR_USERNAME/brendaogutu/sw-en-translation-v1"
+tokenizer = MarianTokenizer.from_pretrained(model_name)
+model = MarianMTModel.from_pretrained(model_name)
+# Translate
+swahili_text = "Habari za asubuhi. Ninaitwa Amina na nina miaka 14."
+inputs = tokenizer(swahili_text, return_tensors="pt", padding=True)
+outputs = model.generate(**inputs, num_beams=5, max_length=256)
+translation = tokenizer.decode(outputs[0], skip_special_tokens=True)
+print(translation)
+```
+## Training Details
+**Base Model:** Helsinki-NLP/opus-mt-mul-en
+**Training Epochs:** 8
+**Batch Size:** 330 (effective: 330)
+**Learning Rate:** 3e-05
+**Optimizer:** AdamW
+**Hardware:** NVIDIA GPU with FP16 mixed precision
+## Evaluation Methodology
+1. **Test Set:** Random 5% split from training distribution
+2. **Domain Evaluation:** Held-out set of real helpline call transcriptions (10-min calls)
+3. **Metrics:** BLEU, chrF, COMET-QE (quality estimation)
+## Citation
+```bibtex
+@software{openchs_translation_2025,
+  author = {BITZ IT Consulting Ltd},
+  title = {Swahili-English Translation Model for OpenCHS},
+  year = {2025},
+  publisher = {HuggingFace},
+  url = {https://huggingface.co/YOUR_USERNAME/brendaogutu/sw-en-translation-v1}
+}
+```
+## Contact
+For questions or issues, please contact: brenda@openchs.org
+---
+*This model is part of the OpenCHS project supporting child helpline services across East Africa.*

baseline_results.json ADDED Viewed

	@@ -0,0 +1,15 @@

+{
+  "eval_loss": 18.69961166381836,
+  "eval_model_preparation_time": 0.0024,
+  "eval_bleu": 0.03199810773966101,
+  "eval_chrf": 20.095068987858355,
+  "eval_runtime": 18.3575,
+  "eval_samples_per_second": 19.611,
+  "eval_steps_per_second": 0.109,
+  "eval_comet": 0.0341746505332392,
+  "eval_comet_std": 0.1127765594779826,
+  "baseline_domain_bleu": 0.0,
+  "baseline_domain_chrf": 1.0286747889121217,
+  "baseline_domain_comet": 4.2171095265075566e-05,
+  "baseline_domain_comet_std": 1.8201333160462296e-07
+}

checkpoint-152/config.json ADDED Viewed

	@@ -0,0 +1,54 @@

+{
+  "activation_dropout": 0.0,
+  "activation_function": "swish",
+  "add_bias_logits": false,
+  "add_final_layer_norm": false,
+  "architectures": [
+    "MarianMTModel"
+  ],
+  "attention_dropout": 0.0,
+  "classif_dropout": 0.0,
+  "classifier_dropout": 0.0,
+  "d_model": 512,
+  "decoder_attention_heads": 8,
+  "decoder_ffn_dim": 2048,
+  "decoder_layerdrop": 0.0,
+  "decoder_layers": 6,
+  "decoder_start_token_id": 64171,
+  "decoder_vocab_size": 64172,
+  "dropout": 0.1,
+  "dtype": "float32",
+  "encoder_attention_heads": 8,
+  "encoder_ffn_dim": 2048,
+  "encoder_layerdrop": 0.0,
+  "encoder_layers": 6,
+  "eos_token_id": 0,
+  "extra_pos_embeddings": 64172,
+  "forced_eos_token_id": 0,
+  "id2label": {
+    "0": "LABEL_0",
+    "1": "LABEL_1",
+    "2": "LABEL_2"
+  },
+  "init_std": 0.02,
+  "is_encoder_decoder": true,
+  "label2id": {
+    "LABEL_0": 0,
+    "LABEL_1": 1,
+    "LABEL_2": 2
+  },
+  "max_length": null,
+  "max_position_embeddings": 512,
+  "model_type": "marian",
+  "normalize_before": false,
+  "normalize_embedding": false,
+  "num_beams": null,
+  "num_hidden_layers": 6,
+  "pad_token_id": 64171,
+  "scale_embedding": true,
+  "share_encoder_decoder_embeddings": true,
+  "static_position_embeddings": true,
+  "transformers_version": "4.57.1",
+  "use_cache": true,
+  "vocab_size": 64172
+}

checkpoint-152/generation_config.json ADDED Viewed

	@@ -0,0 +1,17 @@

+{
+  "bad_words_ids": [
+    [
+      64171
+    ]
+  ],
+  "decoder_start_token_id": 64171,
+  "eos_token_id": [
+    0
+  ],
+  "forced_eos_token_id": 0,
+  "max_length": 512,
+  "num_beams": 6,
+  "pad_token_id": 64171,
+  "renormalize_logits": true,
+  "transformers_version": "4.57.1"
+}

checkpoint-152/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:3d90c9b028acf77d356fb498689d67e8dd01efda8fa45b4a52202dbce37f7951
+size 308263984

checkpoint-152/optimizer.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:efa3fbb0e2f24d9dc12c6ff99b99f5ec2070e436c3d88afd720642d551e90f8a
+size 616171979

checkpoint-152/rng_state.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:44c582f53e45d3dd0b0468a21e55067ed03185392436ddbc3222317fbe93b19d
+size 14645

checkpoint-152/scaler.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:442f2c3e0c6643979ee217f675a17b62b4da04acaf618dae60f841791e71fbc5
+size 1383

checkpoint-152/scheduler.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:2ecfed674202433c3ba86814b5950af3638e3f10c2debedc3aa94709ed01cdbe
+size 1465

checkpoint-152/source.spm ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c4a99ea3602b29fbf901ade8b93a45efa3d7c64eab8fc5fa812383efa327a87d
+size 706917

checkpoint-152/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,5 @@

+{
+  "eos_token": "</s>",
+  "pad_token": "<pad>",
+  "unk_token": "<unk>"
+}

checkpoint-152/target.spm ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c6dce5fa58fcd7dde9e81e279b8c075bf42ee558278f73d6fb48e342029d7f19
+size 791194

checkpoint-152/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,39 @@

+{
+  "added_tokens_decoder": {
+    "0": {
+      "content": "</s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "<unk>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "64171": {
+      "content": "<pad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "</s>",
+  "extra_special_tokens": {},
+  "model_max_length": 512,
+  "pad_token": "<pad>",
+  "separate_vocabs": false,
+  "source_lang": "mul",
+  "sp_model_kwargs": {},
+  "target_lang": "eng",
+  "tokenizer_class": "MarianTokenizer",
+  "unk_token": "<unk>"
+}

checkpoint-152/trainer_state.json ADDED Viewed

	@@ -0,0 +1,64 @@

+{
+  "best_global_step": null,
+  "best_metric": null,
+  "best_model_checkpoint": null,
+  "epoch": 8.0,
+  "eval_steps": 1000,
+  "global_step": 152,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "epoch": 2.6315789473684212,
+      "grad_norm": 2.806515693664551,
+      "learning_rate": 2.584876209978105e-05,
+      "loss": 6.7909,
+      "step": 50
+    },
+    {
+      "epoch": 5.2631578947368425,
+      "grad_norm": 2.533644914627075,
+      "learning_rate": 9.905892790250079e-06,
+      "loss": 3.4474,
+      "step": 100
+    },
+    {
+      "epoch": 7.894736842105263,
+      "grad_norm": 2.563660144805908,
+      "learning_rate": 3.6004094044246314e-08,
+      "loss": 3.2179,
+      "step": 150
+    }
+  ],
+  "logging_steps": 50,
+  "max_steps": 152,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 8,
+  "save_steps": 1000,
+  "stateful_callbacks": {
+    "EarlyStoppingCallback": {
+      "args": {
+        "early_stopping_patience": 5,
+        "early_stopping_threshold": 0.001
+      },
+      "attributes": {
+        "early_stopping_patience_counter": 0
+      }
+    },
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": true,
+        "should_training_stop": true
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 1138408789377024.0,
+  "train_batch_size": 330,
+  "trial_name": null,
+  "trial_params": null
+}

checkpoint-152/training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:eaf461ec2530d2bc87154a757b8eee105da2d899e5550831a3cb068345247e5a
+size 5969

checkpoint-152/vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff

checkpoint-168/config.json ADDED Viewed

	@@ -0,0 +1,54 @@

+{
+  "activation_dropout": 0.0,
+  "activation_function": "swish",
+  "add_bias_logits": false,
+  "add_final_layer_norm": false,
+  "architectures": [
+    "MarianMTModel"
+  ],
+  "attention_dropout": 0.0,
+  "classif_dropout": 0.0,
+  "classifier_dropout": 0.0,
+  "d_model": 512,
+  "decoder_attention_heads": 8,
+  "decoder_ffn_dim": 2048,
+  "decoder_layerdrop": 0.0,
+  "decoder_layers": 6,
+  "decoder_start_token_id": 64171,
+  "decoder_vocab_size": 64172,
+  "dropout": 0.1,
+  "dtype": "float32",
+  "encoder_attention_heads": 8,
+  "encoder_ffn_dim": 2048,
+  "encoder_layerdrop": 0.0,
+  "encoder_layers": 6,
+  "eos_token_id": 0,
+  "extra_pos_embeddings": 64172,
+  "forced_eos_token_id": 0,
+  "id2label": {
+    "0": "LABEL_0",
+    "1": "LABEL_1",
+    "2": "LABEL_2"
+  },
+  "init_std": 0.02,
+  "is_encoder_decoder": true,
+  "label2id": {
+    "LABEL_0": 0,
+    "LABEL_1": 1,
+    "LABEL_2": 2
+  },
+  "max_length": null,
+  "max_position_embeddings": 512,
+  "model_type": "marian",
+  "normalize_before": false,
+  "normalize_embedding": false,
+  "num_beams": null,
+  "num_hidden_layers": 6,
+  "pad_token_id": 64171,
+  "scale_embedding": true,
+  "share_encoder_decoder_embeddings": true,
+  "static_position_embeddings": true,
+  "transformers_version": "4.57.1",
+  "use_cache": true,
+  "vocab_size": 64172
+}

checkpoint-168/generation_config.json ADDED Viewed

	@@ -0,0 +1,17 @@

+{
+  "bad_words_ids": [
+    [
+      64171
+    ]
+  ],
+  "decoder_start_token_id": 64171,
+  "eos_token_id": [
+    0
+  ],
+  "forced_eos_token_id": 0,
+  "max_length": 512,
+  "num_beams": 6,
+  "pad_token_id": 64171,
+  "renormalize_logits": true,
+  "transformers_version": "4.57.1"
+}

checkpoint-168/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:9f963e0d09b97f61f283bb3eab49a67e2529bd1a07b16c991d4ec340c3054c5f
+size 308263984

checkpoint-168/optimizer.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:16d42873865177bd456f9e32244fff31f720603aa14852b2f93703a3a25af3a4
+size 616171979

checkpoint-168/rng_state.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:a24a8728b3b6d13b379536b172d3cc1ad232be65a4814cb20bd4b45a2ebec203
+size 14645

checkpoint-168/scaler.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:ae7ac90dbc48d9822977b941f6c115d423a0663b7a92d188998f39c9e271603c
+size 1383

checkpoint-168/scheduler.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:fb5472e35524a51eecbe54812628f2cd61b41eea270286889a70c444ec421fff
+size 1465

checkpoint-168/source.spm ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c4a99ea3602b29fbf901ade8b93a45efa3d7c64eab8fc5fa812383efa327a87d
+size 706917

checkpoint-168/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,5 @@

+{
+  "eos_token": "</s>",
+  "pad_token": "<pad>",
+  "unk_token": "<unk>"
+}

checkpoint-168/target.spm ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c6dce5fa58fcd7dde9e81e279b8c075bf42ee558278f73d6fb48e342029d7f19
+size 791194

checkpoint-168/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,39 @@

+{
+  "added_tokens_decoder": {
+    "0": {
+      "content": "</s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "<unk>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "64171": {
+      "content": "<pad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "</s>",
+  "extra_special_tokens": {},
+  "model_max_length": 512,
+  "pad_token": "<pad>",
+  "separate_vocabs": false,
+  "source_lang": "mul",
+  "sp_model_kwargs": {},
+  "target_lang": "eng",
+  "tokenizer_class": "MarianTokenizer",
+  "unk_token": "<unk>"
+}

checkpoint-168/trainer_state.json ADDED Viewed

	@@ -0,0 +1,64 @@

+{
+  "best_global_step": null,
+  "best_metric": null,
+  "best_model_checkpoint": null,
+  "epoch": 8.0,
+  "eval_steps": 1000,
+  "global_step": 168,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "epoch": 2.380952380952381,
+      "grad_norm": 2.927729606628418,
+      "learning_rate": 2.6796639964643306e-05,
+      "loss": 6.8435,
+      "step": 50
+    },
+    {
+      "epoch": 4.761904761904762,
+      "grad_norm": 2.7953412532806396,
+      "learning_rate": 1.2977665530323568e-05,
+      "loss": 3.4512,
+      "step": 100
+    },
+    {
+      "epoch": 7.142857142857143,
+      "grad_norm": 2.604569911956787,
+      "learning_rate": 1.1567822802585436e-06,
+      "loss": 3.1858,
+      "step": 150
+    }
+  ],
+  "logging_steps": 50,
+  "max_steps": 168,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 8,
+  "save_steps": 1000,
+  "stateful_callbacks": {
+    "EarlyStoppingCallback": {
+      "args": {
+        "early_stopping_patience": 5,
+        "early_stopping_threshold": 0.001
+      },
+      "attributes": {
+        "early_stopping_patience_counter": 0
+      }
+    },
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": true,
+        "should_training_stop": true
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 1138408789377024.0,
+  "train_batch_size": 300,
+  "trial_name": null,
+  "trial_params": null
+}

checkpoint-168/training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:a8888b7222ef60700b6629d267ff729b6e9f9f8ef70396af3579d5adff82a45e
+size 5969

checkpoint-168/vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff

checkpoint-192/config.json ADDED Viewed

	@@ -0,0 +1,54 @@

+{
+  "activation_dropout": 0.0,
+  "activation_function": "swish",
+  "add_bias_logits": false,
+  "add_final_layer_norm": false,
+  "architectures": [
+    "MarianMTModel"
+  ],
+  "attention_dropout": 0.0,
+  "classif_dropout": 0.0,
+  "classifier_dropout": 0.0,
+  "d_model": 512,
+  "decoder_attention_heads": 8,
+  "decoder_ffn_dim": 2048,
+  "decoder_layerdrop": 0.0,
+  "decoder_layers": 6,
+  "decoder_start_token_id": 64171,
+  "decoder_vocab_size": 64172,
+  "dropout": 0.1,
+  "dtype": "float32",
+  "encoder_attention_heads": 8,
+  "encoder_ffn_dim": 2048,
+  "encoder_layerdrop": 0.0,
+  "encoder_layers": 6,
+  "eos_token_id": 0,
+  "extra_pos_embeddings": 64172,
+  "forced_eos_token_id": 0,
+  "id2label": {
+    "0": "LABEL_0",
+    "1": "LABEL_1",
+    "2": "LABEL_2"
+  },
+  "init_std": 0.02,
+  "is_encoder_decoder": true,
+  "label2id": {
+    "LABEL_0": 0,
+    "LABEL_1": 1,
+    "LABEL_2": 2
+  },
+  "max_length": null,
+  "max_position_embeddings": 512,
+  "model_type": "marian",
+  "normalize_before": false,
+  "normalize_embedding": false,
+  "num_beams": null,
+  "num_hidden_layers": 6,
+  "pad_token_id": 64171,
+  "scale_embedding": true,
+  "share_encoder_decoder_embeddings": true,
+  "static_position_embeddings": true,
+  "transformers_version": "4.57.1",
+  "use_cache": true,
+  "vocab_size": 64172
+}

checkpoint-192/generation_config.json ADDED Viewed

	@@ -0,0 +1,17 @@

+{
+  "bad_words_ids": [
+    [
+      64171
+    ]
+  ],
+  "decoder_start_token_id": 64171,
+  "eos_token_id": [
+    0
+  ],
+  "forced_eos_token_id": 0,
+  "max_length": 512,
+  "num_beams": 6,
+  "pad_token_id": 64171,
+  "renormalize_logits": true,
+  "transformers_version": "4.57.1"
+}

checkpoint-192/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:5234c5bab073fec0655ec089639f5f1b3d5f282b31a368101aa00e5d640cb415
+size 308263984

checkpoint-192/optimizer.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:63e8ba946ad09cd70b8ba7767e18552048c5f7b11351d287405f4b43e4a134e3
+size 616171979

checkpoint-192/rng_state.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:6a23abc4d187d7721fa288f2eae5efa0bd13ff0d80c43eb2578b4f04c0747017
+size 14645

checkpoint-192/scaler.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:cecd44a397927870b6da9d29a983d61c051035d0cdb4ce26ca28a1b4531ad7c2
+size 1383

checkpoint-192/scheduler.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:9e58a2bbcd622aedd411dbce4ac9f21322d7e653d654ce4953c6c4a795539571
+size 1465

checkpoint-192/source.spm ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c4a99ea3602b29fbf901ade8b93a45efa3d7c64eab8fc5fa812383efa327a87d
+size 706917

checkpoint-192/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,5 @@

+{
+  "eos_token": "</s>",
+  "pad_token": "<pad>",
+  "unk_token": "<unk>"
+}

checkpoint-192/target.spm ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c6dce5fa58fcd7dde9e81e279b8c075bf42ee558278f73d6fb48e342029d7f19
+size 791194

checkpoint-192/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,39 @@

+{
+  "added_tokens_decoder": {
+    "0": {
+      "content": "</s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "<unk>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "64171": {
+      "content": "<pad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "</s>",
+  "extra_special_tokens": {},
+  "model_max_length": 512,
+  "pad_token": "<pad>",
+  "separate_vocabs": false,
+  "source_lang": "mul",
+  "sp_model_kwargs": {},
+  "target_lang": "eng",
+  "tokenizer_class": "MarianTokenizer",
+  "unk_token": "<unk>"
+}

checkpoint-192/trainer_state.json ADDED Viewed

	@@ -0,0 +1,64 @@

+{
+  "best_global_step": null,
+  "best_metric": null,
+  "best_model_checkpoint": null,
+  "epoch": 8.0,
+  "eval_steps": 1000,
+  "global_step": 192,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "epoch": 2.0833333333333335,
+      "grad_norm": 3.1876933574676514,
+      "learning_rate": 2.794447789131517e-05,
+      "loss": 6.9797,
+      "step": 50
+    },
+    {
+      "epoch": 4.166666666666667,
+      "grad_norm": 3.0499606132507324,
+      "learning_rate": 1.691261184798162e-05,
+      "loss": 3.4736,
+      "step": 100
+    },
+    {
+      "epoch": 6.25,
+      "grad_norm": 2.7588396072387695,
+      "learning_rate": 4.393398282201788e-06,
+      "loss": 3.1606,
+      "step": 150
+    }
+  ],
+  "logging_steps": 50,
+  "max_steps": 192,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 8,
+  "save_steps": 1000,
+  "stateful_callbacks": {
+    "EarlyStoppingCallback": {
+      "args": {
+        "early_stopping_patience": 5,
+        "early_stopping_threshold": 0.001
+      },
+      "attributes": {
+        "early_stopping_patience_counter": 0
+      }
+    },
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": true,
+        "should_training_stop": true
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 1138408789377024.0,
+  "train_batch_size": 256,
+  "trial_name": null,
+  "trial_params": null
+}

checkpoint-192/training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:752a837b68c2b1da408cae5fbaeab528fc263386b57a2456402bc012c8c6aabf
+size 5969

checkpoint-192/vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff

config.json ADDED Viewed

	@@ -0,0 +1,54 @@

+{
+  "activation_dropout": 0.0,
+  "activation_function": "swish",
+  "add_bias_logits": false,
+  "add_final_layer_norm": false,
+  "architectures": [
+    "MarianMTModel"
+  ],
+  "attention_dropout": 0.0,
+  "classif_dropout": 0.0,
+  "classifier_dropout": 0.0,
+  "d_model": 512,
+  "decoder_attention_heads": 8,
+  "decoder_ffn_dim": 2048,
+  "decoder_layerdrop": 0.0,
+  "decoder_layers": 6,
+  "decoder_start_token_id": 64171,
+  "decoder_vocab_size": 64172,
+  "dropout": 0.1,
+  "dtype": "float32",
+  "encoder_attention_heads": 8,
+  "encoder_ffn_dim": 2048,
+  "encoder_layerdrop": 0.0,
+  "encoder_layers": 6,
+  "eos_token_id": 0,
+  "extra_pos_embeddings": 64172,
+  "forced_eos_token_id": 0,
+  "id2label": {
+    "0": "LABEL_0",
+    "1": "LABEL_1",
+    "2": "LABEL_2"
+  },
+  "init_std": 0.02,
+  "is_encoder_decoder": true,
+  "label2id": {
+    "LABEL_0": 0,
+    "LABEL_1": 1,
+    "LABEL_2": 2
+  },
+  "max_length": null,
+  "max_position_embeddings": 512,
+  "model_type": "marian",
+  "normalize_before": false,
+  "normalize_embedding": false,
+  "num_beams": null,
+  "num_hidden_layers": 6,
+  "pad_token_id": 64171,
+  "scale_embedding": true,
+  "share_encoder_decoder_embeddings": true,
+  "static_position_embeddings": true,
+  "transformers_version": "4.57.1",
+  "use_cache": true,
+  "vocab_size": 64172
+}

generation_config.json ADDED Viewed

	@@ -0,0 +1,17 @@

+{
+  "bad_words_ids": [
+    [
+      64171
+    ]
+  ],
+  "decoder_start_token_id": 64171,
+  "eos_token_id": [
+    0
+  ],
+  "forced_eos_token_id": 0,
+  "max_length": 512,
+  "num_beams": 6,
+  "pad_token_id": 64171,
+  "renormalize_logits": true,
+  "transformers_version": "4.57.1"
+}

model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:3d90c9b028acf77d356fb498689d67e8dd01efda8fa45b4a52202dbce37f7951
+size 308263984

source.spm ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c4a99ea3602b29fbf901ade8b93a45efa3d7c64eab8fc5fa812383efa327a87d
+size 706917

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,5 @@

+{
+  "eos_token": "</s>",
+  "pad_token": "<pad>",
+  "unk_token": "<unk>"
+}