LauraLaureus commited on May 4

Commit

cd00809

verified ·

1 Parent(s): 02c5114

Upload folder using huggingface_hub

Browse files

Files changed (43) hide show

.gitattributes +3 -0
checkpoint-148/1_Pooling/config.json +10 -0
checkpoint-148/README.md +425 -0
checkpoint-148/config.json +27 -0
checkpoint-148/config_sentence_transformers.json +10 -0
checkpoint-148/model.safetensors +3 -0
checkpoint-148/modules.json +20 -0
checkpoint-148/optimizer.pt +3 -0
checkpoint-148/rng_state.pth +3 -0
checkpoint-148/scheduler.pt +3 -0
checkpoint-148/sentence_bert_config.json +4 -0
checkpoint-148/special_tokens_map.json +51 -0
checkpoint-148/tokenizer.json +3 -0
checkpoint-148/tokenizer_config.json +55 -0
checkpoint-148/trainer_state.json +537 -0
checkpoint-148/training_args.bin +3 -0
checkpoint-20/1_Pooling/config.json +10 -0
checkpoint-20/README.md +400 -0
checkpoint-20/config.json +27 -0
checkpoint-20/config_sentence_transformers.json +10 -0
checkpoint-20/model.safetensors +3 -0
checkpoint-20/modules.json +20 -0
checkpoint-20/optimizer.pt +3 -0
checkpoint-20/rng_state.pth +3 -0
checkpoint-20/scheduler.pt +3 -0
checkpoint-20/sentence_bert_config.json +4 -0
checkpoint-20/special_tokens_map.json +51 -0
checkpoint-20/tokenizer.json +3 -0
checkpoint-20/tokenizer_config.json +55 -0
checkpoint-20/trainer_state.json +112 -0
checkpoint-20/training_args.bin +3 -0
eval/similarity_evaluation_results.csv +39 -0
latest/1_Pooling/config.json +10 -0
latest/README.md +404 -0
latest/config.json +27 -0
latest/config_sentence_transformers.json +10 -0
latest/model.safetensors +3 -0
latest/modules.json +20 -0
latest/sentence_bert_config.json +4 -0
latest/special_tokens_map.json +51 -0
latest/tokenizer.json +3 -0
latest/tokenizer_config.json +55 -0
latest/training_args.bin +3 -0

.gitattributes CHANGED Viewed

@@ -33,3 +33,6 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text

 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text
+checkpoint-148/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-20/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+latest/tokenizer.json filter=lfs diff=lfs merge=lfs -text

checkpoint-148/1_Pooling/config.json ADDED Viewed

	@@ -0,0 +1,10 @@

+{
+  "word_embedding_dimension": 1024,
+  "pooling_mode_cls_token": false,
+  "pooling_mode_mean_tokens": true,
+  "pooling_mode_max_tokens": false,
+  "pooling_mode_mean_sqrt_len_tokens": false,
+  "pooling_mode_weightedmean_tokens": false,
+  "pooling_mode_lasttoken": false,
+  "include_prompt": true
+}

checkpoint-148/README.md ADDED Viewed

	@@ -0,0 +1,425 @@

+---
+tags:
+- sentence-transformers
+- sentence-similarity
+- feature-extraction
+- generated_from_trainer
+- dataset_size:290
+- loss:OnlineContrastiveLoss
+base_model: intfloat/multilingual-e5-large
+widget:
+- source_sentence: Antes se coge al mentiroso que al cojo
+  sentences:
+  - A escudero pobre, taza de plata y cántaro de cobre
+  - En río revuelto, pesca abundante
+  - Se ayuda primero al necesitado que al engañador.
+- source_sentence: Asno de muchos, lobos lo comen
+  sentences:
+  - Sabio entre sabios, amigos lo respetan.
+  - El que mucho madruga más hace que el que Dios ayuda.
+  - Se pilla antes a un mentiroso que a un cojo
+- source_sentence: Al buey por el asta, y al hombre por la palabra
+  sentences:
+  - Si no quieres arroz con leche, toma tres tazas
+  - Al hombre por la palabra, y al buey por el cuerno ata
+  - Ese no es tu amigo, sino alguien que siempre busca estar rodeado de bullicio y
+    actividad.
+- source_sentence: Al médico, confesor y letrado, hablarles claro
+  sentences:
+  - Al médico, confesor y letrado, no le hayas engañado
+  - Más vale a quien Dios ayuda que quien mucho madruga
+  - Al que anda entre la miel, algo se le pega
+- source_sentence: A muertos y a idos, no hay amigos
+  sentences:
+  - Al buen callar llaman santo
+  - A los vivos y presentes, siempre hay amigos.
+  - Al que de prestado se viste, en la calle lo desnudan
+pipeline_tag: sentence-similarity
+library_name: sentence-transformers
+metrics:
+- pearson_cosine
+- spearman_cosine
+model-index:
+- name: SentenceTransformer based on intfloat/multilingual-e5-large
+  results:
+  - task:
+      type: semantic-similarity
+      name: Semantic Similarity
+    dataset:
+      name: Unknown
+      type: unknown
+    metrics:
+    - type: pearson_cosine
+      value: 0.8334934833047165
+      name: Pearson Cosine
+    - type: spearman_cosine
+      value: 0.8261353280714282
+      name: Spearman Cosine
+---
+# SentenceTransformer based on intfloat/multilingual-e5-large
+This is a [sentence-transformers](https://www.SBERT.net) model finetuned from [intfloat/multilingual-e5-large](https://huggingface.co/intfloat/multilingual-e5-large) on the csv dataset. It maps sentences & paragraphs to a 1024-dimensional dense vector space and can be used for semantic textual similarity, semantic search, paraphrase mining, text classification, clustering, and more.
+## Model Details
+### Model Description
+- **Model Type:** Sentence Transformer
+- **Base model:** [intfloat/multilingual-e5-large](https://huggingface.co/intfloat/multilingual-e5-large) <!-- at revision 0dc5580a448e4284468b8909bae50fa925907bc5 -->
+- **Maximum Sequence Length:** 512 tokens
+- **Output Dimensionality:** 1024 dimensions
+- **Similarity Function:** Cosine Similarity
+- **Training Dataset:**
+    - csv
+<!-- - **Language:** Unknown -->
+<!-- - **License:** Unknown -->
+### Model Sources
+- **Documentation:** [Sentence Transformers Documentation](https://sbert.net)
+- **Repository:** [Sentence Transformers on GitHub](https://github.com/UKPLab/sentence-transformers)
+- **Hugging Face:** [Sentence Transformers on Hugging Face](https://huggingface.co/models?library=sentence-transformers)
+### Full Model Architecture
+```
+SentenceTransformer(
+  (0): Transformer({'max_seq_length': 512, 'do_lower_case': False}) with Transformer model: XLMRobertaModel
+  (1): Pooling({'word_embedding_dimension': 1024, 'pooling_mode_cls_token': False, 'pooling_mode_mean_tokens': True, 'pooling_mode_max_tokens': False, 'pooling_mode_mean_sqrt_len_tokens': False, 'pooling_mode_weightedmean_tokens': False, 'pooling_mode_lasttoken': False, 'include_prompt': True})
+  (2): Normalize()
+)
+```
+## Usage
+### Direct Usage (Sentence Transformers)
+First install the Sentence Transformers library:
+```bash
+pip install -U sentence-transformers
+```
+Then you can load this model and run inference.
+```python
+from sentence_transformers import SentenceTransformer
+# Download from the 🤗 Hub
+model = SentenceTransformer("sentence_transformers_model_id")
+# Run inference
+sentences = [
+    'A muertos y a idos, no hay amigos',
+    'A los vivos y presentes, siempre hay amigos.',
+    'Al buen callar llaman santo',
+]
+embeddings = model.encode(sentences)
+print(embeddings.shape)
+# [3, 1024]
+# Get the similarity scores for the embeddings
+similarities = model.similarity(embeddings, embeddings)
+print(similarities.shape)
+# [3, 3]
+```
+<!--
+### Direct Usage (Transformers)
+<details><summary>Click to see the direct usage in Transformers</summary>
+</details>
+-->
+<!--
+### Downstream Usage (Sentence Transformers)
+You can finetune this model on your own dataset.
+<details><summary>Click to expand</summary>
+</details>
+-->
+<!--
+### Out-of-Scope Use
+*List how the model may foreseeably be misused and address what users ought not to do with the model.*
+-->
+## Evaluation
+### Metrics
+#### Semantic Similarity
+* Evaluated with [<code>EmbeddingSimilarityEvaluator</code>](https://sbert.net/docs/package_reference/sentence_transformer/evaluation.html#sentence_transformers.evaluation.EmbeddingSimilarityEvaluator)
+| Metric              | Value      |
+|:--------------------|:-----------|
+| pearson_cosine      | 0.8335     |
+| **spearman_cosine** | **0.8261** |
+<!--
+## Bias, Risks and Limitations
+*What are the known or foreseeable issues stemming from this model? You could also flag here known failure cases or weaknesses of the model.*
+-->
+<!--
+### Recommendations
+*What are recommendations with respect to the foreseeable issues? For example, filtering explicit content.*
+-->
+## Training Details
+### Training Dataset
+#### csv
+* Dataset: csv
+* Size: 290 training samples
+* Columns: <code>sentence1</code>, <code>sentence2</code>, and <code>label</code>
+* Approximate statistics based on the first 290 samples:
+  |         | sentence1                                                                         | sentence2                                                                         | label                                           |
+  |:--------|:----------------------------------------------------------------------------------|:----------------------------------------------------------------------------------|:------------------------------------------------|
+  | type    | string                                                                            | string                                                                            | int                                             |
+  | details | <ul><li>min: 7 tokens</li><li>mean: 11.68 tokens</li><li>max: 22 tokens</li></ul> | <ul><li>min: 7 tokens</li><li>mean: 17.01 tokens</li><li>max: 44 tokens</li></ul> | <ul><li>0: ~50.00%</li><li>1: ~50.00%</li></ul> |
+* Samples:
+  | sentence1                                                   | sentence2                                                                                         | label          |
+  |:------------------------------------------------------------|:--------------------------------------------------------------------------------------------------|:---------------|
+  | <code>Gota a gota, la mar se agota.</code>                  | <code>Con el pasar del tiempo se llega a alcanzar cualquier meta.</code>                          | <code>1</code> |
+  | <code>Dime de qué presumes y te diré de qué careces.</code> | <code>Dime de qué careces y te diré de qué dispones.</code>                                       | <code>0</code> |
+  | <code>Cómo se vive, se muere.</code>                        | <code>De aquella forma que hemos vivido nuestra vida será de la forma en la que moriremos.</code> | <code>1</code> |
+* Loss: [<code>OnlineContrastiveLoss</code>](https://sbert.net/docs/package_reference/sentence_transformer/losses.html#onlinecontrastiveloss)
+### Evaluation Dataset
+#### Unnamed Dataset
+* Size: 1,006 evaluation samples
+* Columns: <code>sentence1</code>, <code>sentence2</code>, and <code>label</code>
+* Approximate statistics based on the first 1000 samples:
+  |         | sentence1                                                                         | sentence2                                                                         | label                                           |
+  |:--------|:----------------------------------------------------------------------------------|:----------------------------------------------------------------------------------|:------------------------------------------------|
+  | type    | string                                                                            | string                                                                            | int                                             |
+  | details | <ul><li>min: 7 tokens</li><li>mean: 12.51 tokens</li><li>max: 25 tokens</li></ul> | <ul><li>min: 6 tokens</li><li>mean: 14.82 tokens</li><li>max: 38 tokens</li></ul> | <ul><li>0: ~49.70%</li><li>1: ~50.30%</li></ul> |
+* Samples:
+  | sentence1                                    | sentence2                                                              | label          |
+  |:---------------------------------------------|:-----------------------------------------------------------------------|:---------------|
+  | <code>¿Adónde irá el buey que no are?</code> | <code>¿A dó irá el buey que no are?</code>                             | <code>1</code> |
+  | <code>¿Adónde irá el buey que no are?</code> | <code>¿Adónde irá el buey que no are ni la  mula que no cargue?</code> | <code>1</code> |
+  | <code>¿Adónde irá el buey que no are?</code> | <code>¿Adónde irá el buey que no are, sino al matadero?</code>         | <code>1</code> |
+* Loss: [<code>OnlineContrastiveLoss</code>](https://sbert.net/docs/package_reference/sentence_transformer/losses.html#onlinecontrastiveloss)
+### Training Hyperparameters
+#### Non-Default Hyperparameters
+- `eval_strategy`: steps
+- `learning_rate`: 1e-05
+- `num_train_epochs`: 4
+- `lr_scheduler_type`: constant
+- `load_best_model_at_end`: True
+- `eval_on_start`: True
+- `batch_sampler`: no_duplicates
+#### All Hyperparameters
+<details><summary>Click to expand</summary>
+- `overwrite_output_dir`: False
+- `do_predict`: False
+- `eval_strategy`: steps
+- `prediction_loss_only`: True
+- `per_device_train_batch_size`: 8
+- `per_device_eval_batch_size`: 8
+- `per_gpu_train_batch_size`: None
+- `per_gpu_eval_batch_size`: None
+- `gradient_accumulation_steps`: 1
+- `eval_accumulation_steps`: None
+- `torch_empty_cache_steps`: None
+- `learning_rate`: 1e-05
+- `weight_decay`: 0.0
+- `adam_beta1`: 0.9
+- `adam_beta2`: 0.999
+- `adam_epsilon`: 1e-08
+- `max_grad_norm`: 1.0
+- `num_train_epochs`: 4
+- `max_steps`: -1
+- `lr_scheduler_type`: constant
+- `lr_scheduler_kwargs`: {}
+- `warmup_ratio`: 0.0
+- `warmup_steps`: 0
+- `log_level`: passive
+- `log_level_replica`: warning
+- `log_on_each_node`: True
+- `logging_nan_inf_filter`: True
+- `save_safetensors`: True
+- `save_on_each_node`: False
+- `save_only_model`: False
+- `restore_callback_states_from_checkpoint`: False
+- `no_cuda`: False
+- `use_cpu`: False
+- `use_mps_device`: False
+- `seed`: 42
+- `data_seed`: None
+- `jit_mode_eval`: False
+- `use_ipex`: False
+- `bf16`: False
+- `fp16`: False
+- `fp16_opt_level`: O1
+- `half_precision_backend`: auto
+- `bf16_full_eval`: False
+- `fp16_full_eval`: False
+- `tf32`: None
+- `local_rank`: 0
+- `ddp_backend`: None
+- `tpu_num_cores`: None
+- `tpu_metrics_debug`: False
+- `debug`: []
+- `dataloader_drop_last`: False
+- `dataloader_num_workers`: 0
+- `dataloader_prefetch_factor`: None
+- `past_index`: -1
+- `disable_tqdm`: False
+- `remove_unused_columns`: True
+- `label_names`: None
+- `load_best_model_at_end`: True
+- `ignore_data_skip`: False
+- `fsdp`: []
+- `fsdp_min_num_params`: 0
+- `fsdp_config`: {'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False}
+- `tp_size`: 0
+- `fsdp_transformer_layer_cls_to_wrap`: None
+- `accelerator_config`: {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}
+- `deepspeed`: None
+- `label_smoothing_factor`: 0.0
+- `optim`: adamw_torch
+- `optim_args`: None
+- `adafactor`: False
+- `group_by_length`: False
+- `length_column_name`: length
+- `ddp_find_unused_parameters`: None
+- `ddp_bucket_cap_mb`: None
+- `ddp_broadcast_buffers`: False
+- `dataloader_pin_memory`: True
+- `dataloader_persistent_workers`: False
+- `skip_memory_metrics`: True
+- `use_legacy_prediction_loop`: False
+- `push_to_hub`: False
+- `resume_from_checkpoint`: None
+- `hub_model_id`: None
+- `hub_strategy`: every_save
+- `hub_private_repo`: None
+- `hub_always_push`: False
+- `gradient_checkpointing`: False
+- `gradient_checkpointing_kwargs`: None
+- `include_inputs_for_metrics`: False
+- `include_for_metrics`: []
+- `eval_do_concat_batches`: True
+- `fp16_backend`: auto
+- `push_to_hub_model_id`: None
+- `push_to_hub_organization`: None
+- `mp_parameters`:
+- `auto_find_batch_size`: False
+- `full_determinism`: False
+- `torchdynamo`: None
+- `ray_scope`: last
+- `ddp_timeout`: 1800
+- `torch_compile`: False
+- `torch_compile_backend`: None
+- `torch_compile_mode`: None
+- `dispatch_batches`: None
+- `split_batches`: None
+- `include_tokens_per_second`: False
+- `include_num_input_tokens_seen`: False
+- `neftune_noise_alpha`: None
+- `optim_target_modules`: None
+- `batch_eval_metrics`: False
+- `eval_on_start`: True
+- `use_liger_kernel`: False
+- `eval_use_gather_object`: False
+- `average_tokens_across_devices`: False
+- `prompts`: None
+- `batch_sampler`: no_duplicates
+- `multi_dataset_batch_sampler`: proportional
+</details>
+### Training Logs
+| Epoch  | Step | Training Loss | Validation Loss | spearman_cosine |
+|:------:|:----:|:-------------:|:---------------:|:---------------:|
+| 0      | 0    | -             | 0.1095          | 0.7843          |
+| 0.1351 | 5    | 0.6784        | 0.0765          | 0.8123          |
+| 0.2703 | 10   | 0.5088        | 0.0533          | 0.8303          |
+| 0.4054 | 15   | 0.4364        | 0.0475          | 0.8339          |
+| 0.5405 | 20   | 0.3456        | 0.0435          | 0.8345          |
+| 0.6757 | 25   | 0.1423        | 0.0424          | 0.8324          |
+| 0.8108 | 30   | 0.2852        | 0.0443          | 0.8271          |
+| 0.9459 | 35   | 0.2616        | 0.0514          | 0.8262          |
+| 1.0811 | 40   | 0.1451        | 0.0521          | 0.8232          |
+| 1.2162 | 45   | 0.2046        | 0.0496          | 0.8221          |
+| 1.3514 | 50   | 0.055         | 0.0516          | 0.8197          |
+| 1.4865 | 55   | 0.0956        | 0.0545          | 0.8190          |
+| 1.6216 | 60   | 0.1213        | 0.0533          | 0.8213          |
+| 1.7568 | 65   | 0.2378        | 0.0464          | 0.8253          |
+| 1.8919 | 70   | 0.2723        | 0.0458          | 0.8249          |
+| 2.0270 | 75   | 0.0603        | 0.0467          | 0.8226          |
+| 2.1622 | 80   | 0.1089        | 0.0415          | 0.8263          |
+| 2.2973 | 85   | 0.0813        | 0.0417          | 0.8270          |
+| 2.4324 | 90   | 0.0           | 0.0437          | 0.8250          |
+| 2.5676 | 95   | 0.0436        | 0.0467          | 0.8242          |
+| 2.7027 | 100  | 0.0           | 0.0451          | 0.8242          |
+| 2.8378 | 105  | 0.0           | 0.0451          | 0.8243          |
+| 2.9730 | 110  | 0.0271        | 0.0433          | 0.8243          |
+| 3.1081 | 115  | 0.007         | 0.0502          | 0.8195          |
+| 3.2432 | 120  | 0.1025        | 0.0523          | 0.8195          |
+| 3.3784 | 125  | 0.1244        | 0.0527          | 0.8251          |
+| 3.5135 | 130  | 0.0           | 0.0534          | 0.8262          |
+| 3.6486 | 135  | 0.0259        | 0.0571          | 0.8262          |
+| 3.7838 | 140  | 0.0939        | 0.0526          | 0.8273          |
+| 3.9189 | 145  | 0.1038        | 0.0527          | 0.8261          |
+### Framework Versions
+- Python: 3.12.9
+- Sentence Transformers: 3.4.1
+- Transformers: 4.50.0
+- PyTorch: 2.6.0+cpu
+- Accelerate: 1.6.0
+- Datasets: 3.5.0
+- Tokenizers: 0.21.1
+## Citation
+### BibTeX
+#### Sentence Transformers
+```bibtex
+@inproceedings{reimers-2019-sentence-bert,
+    title = "Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks",
+    author = "Reimers, Nils and Gurevych, Iryna",
+    booktitle = "Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing",
+    month = "11",
+    year = "2019",
+    publisher = "Association for Computational Linguistics",
+    url = "https://arxiv.org/abs/1908.10084",
+}
+```
+<!--
+## Glossary
+*Clearly define terms in order to be accessible across audiences.*
+-->
+<!--
+## Model Card Authors
+*Lists the people who create the model card, providing recognition and accountability for the detailed work that goes into its construction.*
+-->
+<!--
+## Model Card Contact
+*Provides a way for people who have updates to the Model Card, suggestions, or questions, to contact the Model Card authors.*
+-->

checkpoint-148/config.json ADDED Viewed

	@@ -0,0 +1,27 @@

+{
+  "architectures": [
+    "XLMRobertaModel"
+  ],
+  "attention_probs_dropout_prob": 0.1,
+  "bos_token_id": 0,
+  "classifier_dropout": null,
+  "eos_token_id": 2,
+  "hidden_act": "gelu",
+  "hidden_dropout_prob": 0.1,
+  "hidden_size": 1024,
+  "initializer_range": 0.02,
+  "intermediate_size": 4096,
+  "layer_norm_eps": 1e-05,
+  "max_position_embeddings": 514,
+  "model_type": "xlm-roberta",
+  "num_attention_heads": 16,
+  "num_hidden_layers": 24,
+  "output_past": true,
+  "pad_token_id": 1,
+  "position_embedding_type": "absolute",
+  "torch_dtype": "float32",
+  "transformers_version": "4.50.0",
+  "type_vocab_size": 1,
+  "use_cache": true,
+  "vocab_size": 250002
+}

checkpoint-148/config_sentence_transformers.json ADDED Viewed

	@@ -0,0 +1,10 @@

+{
+  "__version__": {
+    "sentence_transformers": "3.4.1",
+    "transformers": "4.50.0",
+    "pytorch": "2.6.0+cpu"
+  },
+  "prompts": {},
+  "default_prompt_name": null,
+  "similarity_fn_name": "cosine"
+}

checkpoint-148/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:184f00394770de2bd5dbb0d1d15cc87926d44ed5251e869dd7a22b49b48a9100
+size 2239607176

checkpoint-148/modules.json ADDED Viewed

	@@ -0,0 +1,20 @@

+[
+  {
+    "idx": 0,
+    "name": "0",
+    "path": "",
+    "type": "sentence_transformers.models.Transformer"
+  },
+  {
+    "idx": 1,
+    "name": "1",
+    "path": "1_Pooling",
+    "type": "sentence_transformers.models.Pooling"
+  },
+  {
+    "idx": 2,
+    "name": "2",
+    "path": "2_Normalize",
+    "type": "sentence_transformers.models.Normalize"
+  }
+]

checkpoint-148/optimizer.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:34abf973396f5c82ed62ced5f3ed810e8fdac393d8282da77c875424ddc22161
+size 4471044921

checkpoint-148/rng_state.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:289acee9b20675fcf33130539b2d32f8fe82f5b8a76ff22e77e5702e32d167f1
+size 13990

checkpoint-148/scheduler.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:146ef9b5e5cd6fc6dfefe2364ebf158e1371c154a444a4d236aaa6ca0953aa1b
+size 1064

checkpoint-148/sentence_bert_config.json ADDED Viewed

	@@ -0,0 +1,4 @@

+{
+  "max_seq_length": 512,
+  "do_lower_case": false
+}

checkpoint-148/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,51 @@

+{
+  "bos_token": {
+    "content": "<s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "cls_token": {
+    "content": "<s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "</s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "mask_token": {
+    "content": "<mask>",
+    "lstrip": true,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "<pad>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "sep_token": {
+    "content": "</s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "<unk>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

checkpoint-148/tokenizer.json ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:883b037111086fd4dfebbbc9b7cee11e1517b5e0c0514879478661440f137085
+size 17082987

checkpoint-148/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,55 @@

+{
+  "added_tokens_decoder": {
+    "0": {
+      "content": "<s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "<pad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "</s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "3": {
+      "content": "<unk>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "250001": {
+      "content": "<mask>",
+      "lstrip": true,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "<s>",
+  "clean_up_tokenization_spaces": true,
+  "cls_token": "<s>",
+  "eos_token": "</s>",
+  "extra_special_tokens": {},
+  "mask_token": "<mask>",
+  "model_max_length": 512,
+  "pad_token": "<pad>",
+  "sep_token": "</s>",
+  "tokenizer_class": "XLMRobertaTokenizer",
+  "unk_token": "<unk>"
+}

checkpoint-148/trainer_state.json ADDED Viewed

	@@ -0,0 +1,537 @@

+{
+  "best_global_step": 80,
+  "best_metric": 0.04149133339524269,
+  "best_model_checkpoint": "models/me5-large-retraining\\checkpoint-80",
+  "epoch": 4.0,
+  "eval_steps": 5,
+  "global_step": 148,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "epoch": 0,
+      "eval_loss": 0.10950354486703873,
+      "eval_pearson_cosine": 0.7806081242807599,
+      "eval_runtime": 154.7907,
+      "eval_samples_per_second": 6.499,
+      "eval_spearman_cosine": 0.7843279594448466,
+      "eval_steps_per_second": 0.814,
+      "step": 0
+    },
+    {
+      "epoch": 0.13513513513513514,
+      "grad_norm": 9.38223934173584,
+      "learning_rate": 1e-05,
+      "loss": 0.6784,
+      "step": 5
+    },
+    {
+      "epoch": 0.13513513513513514,
+      "eval_loss": 0.07647562772035599,
+      "eval_pearson_cosine": 0.8193313583571735,
+      "eval_runtime": 154.4069,
+      "eval_samples_per_second": 6.515,
+      "eval_spearman_cosine": 0.8122999445241028,
+      "eval_steps_per_second": 0.816,
+      "step": 5
+    },
+    {
+      "epoch": 0.2702702702702703,
+      "grad_norm": 8.665059089660645,
+      "learning_rate": 1e-05,
+      "loss": 0.5088,
+      "step": 10
+    },
+    {
+      "epoch": 0.2702702702702703,
+      "eval_loss": 0.05329965054988861,
+      "eval_pearson_cosine": 0.846437709431274,
+      "eval_runtime": 157.2408,
+      "eval_samples_per_second": 6.398,
+      "eval_spearman_cosine": 0.8303112726354404,
+      "eval_steps_per_second": 0.801,
+      "step": 10
+    },
+    {
+      "epoch": 0.40540540540540543,
+      "grad_norm": 21.96602439880371,
+      "learning_rate": 1e-05,
+      "loss": 0.4364,
+      "step": 15
+    },
+    {
+      "epoch": 0.40540540540540543,
+      "eval_loss": 0.047497160732746124,
+      "eval_pearson_cosine": 0.8497791216636505,
+      "eval_runtime": 153.8649,
+      "eval_samples_per_second": 6.538,
+      "eval_spearman_cosine": 0.8338916292060913,
+      "eval_steps_per_second": 0.819,
+      "step": 15
+    },
+    {
+      "epoch": 0.5405405405405406,
+      "grad_norm": 10.698002815246582,
+      "learning_rate": 1e-05,
+      "loss": 0.3456,
+      "step": 20
+    },
+    {
+      "epoch": 0.5405405405405406,
+      "eval_loss": 0.043546345084905624,
+      "eval_pearson_cosine": 0.8481768671521202,
+      "eval_runtime": 154.4013,
+      "eval_samples_per_second": 6.515,
+      "eval_spearman_cosine": 0.8344803713886917,
+      "eval_steps_per_second": 0.816,
+      "step": 20
+    },
+    {
+      "epoch": 0.6756756756756757,
+      "grad_norm": 0.0,
+      "learning_rate": 1e-05,
+      "loss": 0.1423,
+      "step": 25
+    },
+    {
+      "epoch": 0.6756756756756757,
+      "eval_loss": 0.042350731790065765,
+      "eval_pearson_cosine": 0.8419294236114762,
+      "eval_runtime": 153.4308,
+      "eval_samples_per_second": 6.557,
+      "eval_spearman_cosine": 0.8324197823497284,
+      "eval_steps_per_second": 0.821,
+      "step": 25
+    },
+    {
+      "epoch": 0.8108108108108109,
+      "grad_norm": 11.502903938293457,
+      "learning_rate": 1e-05,
+      "loss": 0.2852,
+      "step": 30
+    },
+    {
+      "epoch": 0.8108108108108109,
+      "eval_loss": 0.04431174322962761,
+      "eval_pearson_cosine": 0.8311833878571917,
+      "eval_runtime": 153.1117,
+      "eval_samples_per_second": 6.57,
+      "eval_spearman_cosine": 0.8270800450821792,
+      "eval_steps_per_second": 0.823,
+      "step": 30
+    },
+    {
+      "epoch": 0.9459459459459459,
+      "grad_norm": 13.956666946411133,
+      "learning_rate": 1e-05,
+      "loss": 0.2616,
+      "step": 35
+    },
+    {
+      "epoch": 0.9459459459459459,
+      "eval_loss": 0.05144113302230835,
+      "eval_pearson_cosine": 0.8323883964862309,
+      "eval_runtime": 153.6257,
+      "eval_samples_per_second": 6.548,
+      "eval_spearman_cosine": 0.8261627064456549,
+      "eval_steps_per_second": 0.82,
+      "step": 35
+    },
+    {
+      "epoch": 1.0810810810810811,
+      "grad_norm": 0.0,
+      "learning_rate": 1e-05,
+      "loss": 0.1451,
+      "step": 40
+    },
+    {
+      "epoch": 1.0810810810810811,
+      "eval_loss": 0.052108317613601685,
+      "eval_pearson_cosine": 0.8322596480425719,
+      "eval_runtime": 152.7732,
+      "eval_samples_per_second": 6.585,
+      "eval_spearman_cosine": 0.8232463910787938,
+      "eval_steps_per_second": 0.825,
+      "step": 40
+    },
+    {
+      "epoch": 1.2162162162162162,
+      "grad_norm": 11.031551361083984,
+      "learning_rate": 1e-05,
+      "loss": 0.2046,
+      "step": 45
+    },
+    {
+      "epoch": 1.2162162162162162,
+      "eval_loss": 0.049648430198431015,
+      "eval_pearson_cosine": 0.8345968763719022,
+      "eval_runtime": 151.5419,
+      "eval_samples_per_second": 6.638,
+      "eval_spearman_cosine": 0.8220928743948337,
+      "eval_steps_per_second": 0.831,
+      "step": 45
+    },
+    {
+      "epoch": 1.3513513513513513,
+      "grad_norm": 0.0,
+      "learning_rate": 1e-05,
+      "loss": 0.055,
+      "step": 50
+    },
+    {
+      "epoch": 1.3513513513513513,
+      "eval_loss": 0.05155247077345848,
+      "eval_pearson_cosine": 0.8295312877735412,
+      "eval_runtime": 152.4645,
+      "eval_samples_per_second": 6.598,
+      "eval_spearman_cosine": 0.8197002635410278,
+      "eval_steps_per_second": 0.826,
+      "step": 50
+    },
+    {
+      "epoch": 1.4864864864864864,
+      "grad_norm": 0.0,
+      "learning_rate": 1e-05,
+      "loss": 0.0956,
+      "step": 55
+    },
+    {
+      "epoch": 1.4864864864864864,
+      "eval_loss": 0.05453842505812645,
+      "eval_pearson_cosine": 0.8258612961974536,
+      "eval_runtime": 151.2464,
+      "eval_samples_per_second": 6.651,
+      "eval_spearman_cosine": 0.8190362223126076,
+      "eval_steps_per_second": 0.833,
+      "step": 55
+    },
+    {
+      "epoch": 1.6216216216216215,
+      "grad_norm": 10.073129653930664,
+      "learning_rate": 1e-05,
+      "loss": 0.1213,
+      "step": 60
+    },
+    {
+      "epoch": 1.6216216216216215,
+      "eval_loss": 0.05331311747431755,
+      "eval_pearson_cosine": 0.8280490163439709,
+      "eval_runtime": 151.9002,
+      "eval_samples_per_second": 6.623,
+      "eval_spearman_cosine": 0.8212679517806181,
+      "eval_steps_per_second": 0.829,
+      "step": 60
+    },
+    {
+      "epoch": 1.7567567567567568,
+      "grad_norm": 8.610294342041016,
+      "learning_rate": 1e-05,
+      "loss": 0.2378,
+      "step": 65
+    },
+    {
+      "epoch": 1.7567567567567568,
+      "eval_loss": 0.04638493061065674,
+      "eval_pearson_cosine": 0.8348764332747338,
+      "eval_runtime": 151.3483,
+      "eval_samples_per_second": 6.647,
+      "eval_spearman_cosine": 0.8253343633484949,
+      "eval_steps_per_second": 0.833,
+      "step": 65
+    },
+    {
+      "epoch": 1.8918918918918919,
+      "grad_norm": 7.2265729904174805,
+      "learning_rate": 1e-05,
+      "loss": 0.2723,
+      "step": 70
+    },
+    {
+      "epoch": 1.8918918918918919,
+      "eval_loss": 0.04580773040652275,
+      "eval_pearson_cosine": 0.8327080612027193,
+      "eval_runtime": 151.0951,
+      "eval_samples_per_second": 6.658,
+      "eval_spearman_cosine": 0.8249373111883099,
+      "eval_steps_per_second": 0.834,
+      "step": 70
+    },
+    {
+      "epoch": 2.027027027027027,
+      "grad_norm": 0.0,
+      "learning_rate": 1e-05,
+      "loss": 0.0603,
+      "step": 75
+    },
+    {
+      "epoch": 2.027027027027027,
+      "eval_loss": 0.04667947068810463,
+      "eval_pearson_cosine": 0.8292172905350813,
+      "eval_runtime": 150.817,
+      "eval_samples_per_second": 6.67,
+      "eval_spearman_cosine": 0.8226234223032437,
+      "eval_steps_per_second": 0.835,
+      "step": 75
+    },
+    {
+      "epoch": 2.1621621621621623,
+      "grad_norm": 7.278922080993652,
+      "learning_rate": 1e-05,
+      "loss": 0.1089,
+      "step": 80
+    },
+    {
+      "epoch": 2.1621621621621623,
+      "eval_loss": 0.04149133339524269,
+      "eval_pearson_cosine": 0.8348787263877363,
+      "eval_runtime": 151.9077,
+      "eval_samples_per_second": 6.622,
+      "eval_spearman_cosine": 0.8262517044196894,
+      "eval_steps_per_second": 0.829,
+      "step": 80
+    },
+    {
+      "epoch": 2.2972972972972974,
+      "grad_norm": 0.0,
+      "learning_rate": 1e-05,
+      "loss": 0.0813,
+      "step": 85
+    },
+    {
+      "epoch": 2.2972972972972974,
+      "eval_loss": 0.041691090911626816,
+      "eval_pearson_cosine": 0.834739922043313,
+      "eval_runtime": 151.4069,
+      "eval_samples_per_second": 6.644,
+      "eval_spearman_cosine": 0.8269978977904043,
+      "eval_steps_per_second": 0.832,
+      "step": 85
+    },
+    {
+      "epoch": 2.4324324324324325,
+      "grad_norm": 0.0,
+      "learning_rate": 1e-05,
+      "loss": 0.0,
+      "step": 90
+    },
+    {
+      "epoch": 2.4324324324324325,
+      "eval_loss": 0.043681543320417404,
+      "eval_pearson_cosine": 0.8300953895591171,
+      "eval_runtime": 151.2024,
+      "eval_samples_per_second": 6.653,
+      "eval_spearman_cosine": 0.8249920776743955,
+      "eval_steps_per_second": 0.833,
+      "step": 90
+    },
+    {
+      "epoch": 2.5675675675675675,
+      "grad_norm": 0.0,
+      "learning_rate": 1e-05,
+      "loss": 0.0436,
+      "step": 95
+    },
+    {
+      "epoch": 2.5675675675675675,
+      "eval_loss": 0.04666070267558098,
+      "eval_pearson_cosine": 0.8280036278693699,
+      "eval_runtime": 150.6994,
+      "eval_samples_per_second": 6.676,
+      "eval_spearman_cosine": 0.8241911129581995,
+      "eval_steps_per_second": 0.836,
+      "step": 95
+    },
+    {
+      "epoch": 2.7027027027027026,
+      "grad_norm": 0.0,
+      "learning_rate": 1e-05,
+      "loss": 0.0,
+      "step": 100
+    },
+    {
+      "epoch": 2.7027027027027026,
+      "eval_loss": 0.04513184353709221,
+      "eval_pearson_cosine": 0.827662426497365,
+      "eval_runtime": 151.7254,
+      "eval_samples_per_second": 6.63,
+      "eval_spearman_cosine": 0.8241911153867979,
+      "eval_steps_per_second": 0.83,
+      "step": 100
+    },
+    {
+      "epoch": 2.8378378378378377,
+      "grad_norm": 0.0,
+      "learning_rate": 1e-05,
+      "loss": 0.0,
+      "step": 105
+    },
+    {
+      "epoch": 2.8378378378378377,
+      "eval_loss": 0.045100126415491104,
+      "eval_pearson_cosine": 0.8271686930217653,
+      "eval_runtime": 150.6791,
+      "eval_samples_per_second": 6.676,
+      "eval_spearman_cosine": 0.8242595734942031,
+      "eval_steps_per_second": 0.836,
+      "step": 105
+    },
+    {
+      "epoch": 2.972972972972973,
+      "grad_norm": 0.0,
+      "learning_rate": 1e-05,
+      "loss": 0.0271,
+      "step": 110
+    },
+    {
+      "epoch": 2.972972972972973,
+      "eval_loss": 0.04326998442411423,
+      "eval_pearson_cosine": 0.8242998475213165,
+      "eval_runtime": 151.5644,
+      "eval_samples_per_second": 6.637,
+      "eval_spearman_cosine": 0.8243348749833265,
+      "eval_steps_per_second": 0.831,
+      "step": 110
+    },
+    {
+      "epoch": 3.108108108108108,
+      "grad_norm": 0.0,
+      "learning_rate": 1e-05,
+      "loss": 0.007,
+      "step": 115
+    },
+    {
+      "epoch": 3.108108108108108,
+      "eval_loss": 0.050163887441158295,
+      "eval_pearson_cosine": 0.8100599157021782,
+      "eval_runtime": 149.8939,
+      "eval_samples_per_second": 6.711,
+      "eval_spearman_cosine": 0.8195085832550942,
+      "eval_steps_per_second": 0.841,
+      "step": 115
+    },
+    {
+      "epoch": 3.2432432432432434,
+      "grad_norm": 5.909173011779785,
+      "learning_rate": 1e-05,
+      "loss": 0.1025,
+      "step": 120
+    },
+    {
+      "epoch": 3.2432432432432434,
+      "eval_loss": 0.052336592227220535,
+      "eval_pearson_cosine": 0.8092739374985023,
+      "eval_runtime": 151.2544,
+      "eval_samples_per_second": 6.651,
+      "eval_spearman_cosine": 0.8194743493718913,
+      "eval_steps_per_second": 0.833,
+      "step": 120
+    },
+    {
+      "epoch": 3.3783783783783785,
+      "grad_norm": 0.0,
+      "learning_rate": 1e-05,
+      "loss": 0.1244,
+      "step": 125
+    },
+    {
+      "epoch": 3.3783783783783785,
+      "eval_loss": 0.05269436165690422,
+      "eval_pearson_cosine": 0.8212737827789367,
+      "eval_runtime": 150.843,
+      "eval_samples_per_second": 6.669,
+      "eval_spearman_cosine": 0.8250605382131625,
+      "eval_steps_per_second": 0.835,
+      "step": 125
+    },
+    {
+      "epoch": 3.5135135135135136,
+      "grad_norm": 0.0,
+      "learning_rate": 1e-05,
+      "loss": 0.0,
+      "step": 130
+    },
+    {
+      "epoch": 3.5135135135135136,
+      "eval_loss": 0.05343884229660034,
+      "eval_pearson_cosine": 0.8257663666576973,
+      "eval_runtime": 150.9582,
+      "eval_samples_per_second": 6.664,
+      "eval_spearman_cosine": 0.8261900896885362,
+      "eval_steps_per_second": 0.835,
+      "step": 130
+    },
+    {
+      "epoch": 3.6486486486486487,
+      "grad_norm": 0.0,
+      "learning_rate": 1e-05,
+      "loss": 0.0259,
+      "step": 135
+    },
+    {
+      "epoch": 3.6486486486486487,
+      "eval_loss": 0.05709109827876091,
+      "eval_pearson_cosine": 0.8294106767298322,
+      "eval_runtime": 151.0358,
+      "eval_samples_per_second": 6.661,
+      "eval_spearman_cosine": 0.826217475365987,
+      "eval_steps_per_second": 0.834,
+      "step": 135
+    },
+    {
+      "epoch": 3.7837837837837838,
+      "grad_norm": 0.0,
+      "learning_rate": 1e-05,
+      "loss": 0.0939,
+      "step": 140
+    },
+    {
+      "epoch": 3.7837837837837838,
+      "eval_loss": 0.05264894291758537,
+      "eval_pearson_cosine": 0.8333850683405307,
+      "eval_runtime": 151.069,
+      "eval_samples_per_second": 6.659,
+      "eval_spearman_cosine": 0.8273128026466706,
+      "eval_steps_per_second": 0.834,
+      "step": 140
+    },
+    {
+      "epoch": 3.918918918918919,
+      "grad_norm": 7.900262832641602,
+      "learning_rate": 1e-05,
+      "loss": 0.1038,
+      "step": 145
+    },
+    {
+      "epoch": 3.918918918918919,
+      "eval_loss": 0.0527377724647522,
+      "eval_pearson_cosine": 0.8334934833047165,
+      "eval_runtime": 150.8011,
+      "eval_samples_per_second": 6.671,
+      "eval_spearman_cosine": 0.8261353280714282,
+      "eval_steps_per_second": 0.836,
+      "step": 145
+    }
+  ],
+  "logging_steps": 5,
+  "max_steps": 148,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 4,
+  "save_steps": 10,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": true,
+        "should_training_stop": true
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 0.0,
+  "train_batch_size": 8,
+  "trial_name": null,
+  "trial_params": null
+}

checkpoint-148/training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c6195991b123e7d9abf5d6fe39d2acc574f894288d505cb696178ae95f5cb07f
+size 5624

checkpoint-20/1_Pooling/config.json ADDED Viewed

	@@ -0,0 +1,10 @@

+{
+  "word_embedding_dimension": 1024,
+  "pooling_mode_cls_token": false,
+  "pooling_mode_mean_tokens": true,
+  "pooling_mode_max_tokens": false,
+  "pooling_mode_mean_sqrt_len_tokens": false,
+  "pooling_mode_weightedmean_tokens": false,
+  "pooling_mode_lasttoken": false,
+  "include_prompt": true
+}

checkpoint-20/README.md ADDED Viewed

	@@ -0,0 +1,400 @@

+---
+tags:
+- sentence-transformers
+- sentence-similarity
+- feature-extraction
+- generated_from_trainer
+- dataset_size:290
+- loss:OnlineContrastiveLoss
+base_model: intfloat/multilingual-e5-large
+widget:
+- source_sentence: Antes se coge al mentiroso que al cojo
+  sentences:
+  - A escudero pobre, taza de plata y cántaro de cobre
+  - En río revuelto, pesca abundante
+  - Se ayuda primero al necesitado que al engañador.
+- source_sentence: Asno de muchos, lobos lo comen
+  sentences:
+  - Sabio entre sabios, amigos lo respetan.
+  - El que mucho madruga más hace que el que Dios ayuda.
+  - Se pilla antes a un mentiroso que a un cojo
+- source_sentence: Al buey por el asta, y al hombre por la palabra
+  sentences:
+  - Si no quieres arroz con leche, toma tres tazas
+  - Al hombre por la palabra, y al buey por el cuerno ata
+  - Ese no es tu amigo, sino alguien que siempre busca estar rodeado de bullicio y
+    actividad.
+- source_sentence: Al médico, confesor y letrado, hablarles claro
+  sentences:
+  - Al médico, confesor y letrado, no le hayas engañado
+  - Más vale a quien Dios ayuda que quien mucho madruga
+  - Al que anda entre la miel, algo se le pega
+- source_sentence: A muertos y a idos, no hay amigos
+  sentences:
+  - Al buen callar llaman santo
+  - A los vivos y presentes, siempre hay amigos.
+  - Al que de prestado se viste, en la calle lo desnudan
+pipeline_tag: sentence-similarity
+library_name: sentence-transformers
+metrics:
+- pearson_cosine
+- spearman_cosine
+model-index:
+- name: SentenceTransformer based on intfloat/multilingual-e5-large
+  results:
+  - task:
+      type: semantic-similarity
+      name: Semantic Similarity
+    dataset:
+      name: Unknown
+      type: unknown
+    metrics:
+    - type: pearson_cosine
+      value: 0.8481768671521202
+      name: Pearson Cosine
+    - type: spearman_cosine
+      value: 0.8344803713886917
+      name: Spearman Cosine
+---
+# SentenceTransformer based on intfloat/multilingual-e5-large
+This is a [sentence-transformers](https://www.SBERT.net) model finetuned from [intfloat/multilingual-e5-large](https://huggingface.co/intfloat/multilingual-e5-large) on the csv dataset. It maps sentences & paragraphs to a 1024-dimensional dense vector space and can be used for semantic textual similarity, semantic search, paraphrase mining, text classification, clustering, and more.
+## Model Details
+### Model Description
+- **Model Type:** Sentence Transformer
+- **Base model:** [intfloat/multilingual-e5-large](https://huggingface.co/intfloat/multilingual-e5-large) <!-- at revision 0dc5580a448e4284468b8909bae50fa925907bc5 -->
+- **Maximum Sequence Length:** 512 tokens
+- **Output Dimensionality:** 1024 dimensions
+- **Similarity Function:** Cosine Similarity
+- **Training Dataset:**
+    - csv
+<!-- - **Language:** Unknown -->
+<!-- - **License:** Unknown -->
+### Model Sources
+- **Documentation:** [Sentence Transformers Documentation](https://sbert.net)
+- **Repository:** [Sentence Transformers on GitHub](https://github.com/UKPLab/sentence-transformers)
+- **Hugging Face:** [Sentence Transformers on Hugging Face](https://huggingface.co/models?library=sentence-transformers)
+### Full Model Architecture
+```
+SentenceTransformer(
+  (0): Transformer({'max_seq_length': 512, 'do_lower_case': False}) with Transformer model: XLMRobertaModel
+  (1): Pooling({'word_embedding_dimension': 1024, 'pooling_mode_cls_token': False, 'pooling_mode_mean_tokens': True, 'pooling_mode_max_tokens': False, 'pooling_mode_mean_sqrt_len_tokens': False, 'pooling_mode_weightedmean_tokens': False, 'pooling_mode_lasttoken': False, 'include_prompt': True})
+  (2): Normalize()
+)
+```
+## Usage
+### Direct Usage (Sentence Transformers)
+First install the Sentence Transformers library:
+```bash
+pip install -U sentence-transformers
+```
+Then you can load this model and run inference.
+```python
+from sentence_transformers import SentenceTransformer
+# Download from the 🤗 Hub
+model = SentenceTransformer("sentence_transformers_model_id")
+# Run inference
+sentences = [
+    'A muertos y a idos, no hay amigos',
+    'A los vivos y presentes, siempre hay amigos.',
+    'Al buen callar llaman santo',
+]
+embeddings = model.encode(sentences)
+print(embeddings.shape)
+# [3, 1024]
+# Get the similarity scores for the embeddings
+similarities = model.similarity(embeddings, embeddings)
+print(similarities.shape)
+# [3, 3]
+```
+<!--
+### Direct Usage (Transformers)
+<details><summary>Click to see the direct usage in Transformers</summary>
+</details>
+-->
+<!--
+### Downstream Usage (Sentence Transformers)
+You can finetune this model on your own dataset.
+<details><summary>Click to expand</summary>
+</details>
+-->
+<!--
+### Out-of-Scope Use
+*List how the model may foreseeably be misused and address what users ought not to do with the model.*
+-->
+## Evaluation
+### Metrics
+#### Semantic Similarity
+* Evaluated with [<code>EmbeddingSimilarityEvaluator</code>](https://sbert.net/docs/package_reference/sentence_transformer/evaluation.html#sentence_transformers.evaluation.EmbeddingSimilarityEvaluator)
+| Metric              | Value      |
+|:--------------------|:-----------|
+| pearson_cosine      | 0.8482     |
+| **spearman_cosine** | **0.8345** |
+<!--
+## Bias, Risks and Limitations
+*What are the known or foreseeable issues stemming from this model? You could also flag here known failure cases or weaknesses of the model.*
+-->
+<!--
+### Recommendations
+*What are recommendations with respect to the foreseeable issues? For example, filtering explicit content.*
+-->
+## Training Details
+### Training Dataset
+#### csv
+* Dataset: csv
+* Size: 290 training samples
+* Columns: <code>sentence1</code>, <code>sentence2</code>, and <code>label</code>
+* Approximate statistics based on the first 290 samples:
+  |         | sentence1                                                                         | sentence2                                                                         | label                                           |
+  |:--------|:----------------------------------------------------------------------------------|:----------------------------------------------------------------------------------|:------------------------------------------------|
+  | type    | string                                                                            | string                                                                            | int                                             |
+  | details | <ul><li>min: 7 tokens</li><li>mean: 11.68 tokens</li><li>max: 22 tokens</li></ul> | <ul><li>min: 7 tokens</li><li>mean: 17.01 tokens</li><li>max: 44 tokens</li></ul> | <ul><li>0: ~50.00%</li><li>1: ~50.00%</li></ul> |
+* Samples:
+  | sentence1                                                   | sentence2                                                                                         | label          |
+  |:------------------------------------------------------------|:--------------------------------------------------------------------------------------------------|:---------------|
+  | <code>Gota a gota, la mar se agota.</code>                  | <code>Con el pasar del tiempo se llega a alcanzar cualquier meta.</code>                          | <code>1</code> |
+  | <code>Dime de qué presumes y te diré de qué careces.</code> | <code>Dime de qué careces y te diré de qué dispones.</code>                                       | <code>0</code> |
+  | <code>Cómo se vive, se muere.</code>                        | <code>De aquella forma que hemos vivido nuestra vida será de la forma en la que moriremos.</code> | <code>1</code> |
+* Loss: [<code>OnlineContrastiveLoss</code>](https://sbert.net/docs/package_reference/sentence_transformer/losses.html#onlinecontrastiveloss)
+### Evaluation Dataset
+#### Unnamed Dataset
+* Size: 1,006 evaluation samples
+* Columns: <code>sentence1</code>, <code>sentence2</code>, and <code>label</code>
+* Approximate statistics based on the first 1000 samples:
+  |         | sentence1                                                                         | sentence2                                                                         | label                                           |
+  |:--------|:----------------------------------------------------------------------------------|:----------------------------------------------------------------------------------|:------------------------------------------------|
+  | type    | string                                                                            | string                                                                            | int                                             |
+  | details | <ul><li>min: 7 tokens</li><li>mean: 12.51 tokens</li><li>max: 25 tokens</li></ul> | <ul><li>min: 6 tokens</li><li>mean: 14.82 tokens</li><li>max: 38 tokens</li></ul> | <ul><li>0: ~49.70%</li><li>1: ~50.30%</li></ul> |
+* Samples:
+  | sentence1                                    | sentence2                                                              | label          |
+  |:---------------------------------------------|:-----------------------------------------------------------------------|:---------------|
+  | <code>¿Adónde irá el buey que no are?</code> | <code>¿A dó irá el buey que no are?</code>                             | <code>1</code> |
+  | <code>¿Adónde irá el buey que no are?</code> | <code>¿Adónde irá el buey que no are ni la  mula que no cargue?</code> | <code>1</code> |
+  | <code>¿Adónde irá el buey que no are?</code> | <code>¿Adónde irá el buey que no are, sino al matadero?</code>         | <code>1</code> |
+* Loss: [<code>OnlineContrastiveLoss</code>](https://sbert.net/docs/package_reference/sentence_transformer/losses.html#onlinecontrastiveloss)
+### Training Hyperparameters
+#### Non-Default Hyperparameters
+- `eval_strategy`: steps
+- `learning_rate`: 1e-05
+- `num_train_epochs`: 1
+- `lr_scheduler_type`: constant
+- `load_best_model_at_end`: True
+- `eval_on_start`: True
+- `batch_sampler`: no_duplicates
+#### All Hyperparameters
+<details><summary>Click to expand</summary>
+- `overwrite_output_dir`: False
+- `do_predict`: False
+- `eval_strategy`: steps
+- `prediction_loss_only`: True
+- `per_device_train_batch_size`: 8
+- `per_device_eval_batch_size`: 8
+- `per_gpu_train_batch_size`: None
+- `per_gpu_eval_batch_size`: None
+- `gradient_accumulation_steps`: 1
+- `eval_accumulation_steps`: None
+- `torch_empty_cache_steps`: None
+- `learning_rate`: 1e-05
+- `weight_decay`: 0.0
+- `adam_beta1`: 0.9
+- `adam_beta2`: 0.999
+- `adam_epsilon`: 1e-08
+- `max_grad_norm`: 1.0
+- `num_train_epochs`: 1
+- `max_steps`: -1
+- `lr_scheduler_type`: constant
+- `lr_scheduler_kwargs`: {}
+- `warmup_ratio`: 0.0
+- `warmup_steps`: 0
+- `log_level`: passive
+- `log_level_replica`: warning
+- `log_on_each_node`: True
+- `logging_nan_inf_filter`: True
+- `save_safetensors`: True
+- `save_on_each_node`: False
+- `save_only_model`: False
+- `restore_callback_states_from_checkpoint`: False
+- `no_cuda`: False
+- `use_cpu`: False
+- `use_mps_device`: False
+- `seed`: 42
+- `data_seed`: None
+- `jit_mode_eval`: False
+- `use_ipex`: False
+- `bf16`: False
+- `fp16`: False
+- `fp16_opt_level`: O1
+- `half_precision_backend`: auto
+- `bf16_full_eval`: False
+- `fp16_full_eval`: False
+- `tf32`: None
+- `local_rank`: 0
+- `ddp_backend`: None
+- `tpu_num_cores`: None
+- `tpu_metrics_debug`: False
+- `debug`: []
+- `dataloader_drop_last`: False
+- `dataloader_num_workers`: 0
+- `dataloader_prefetch_factor`: None
+- `past_index`: -1
+- `disable_tqdm`: False
+- `remove_unused_columns`: True
+- `label_names`: None
+- `load_best_model_at_end`: True
+- `ignore_data_skip`: False
+- `fsdp`: []
+- `fsdp_min_num_params`: 0
+- `fsdp_config`: {'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False}
+- `tp_size`: 0
+- `fsdp_transformer_layer_cls_to_wrap`: None
+- `accelerator_config`: {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}
+- `deepspeed`: None
+- `label_smoothing_factor`: 0.0
+- `optim`: adamw_torch
+- `optim_args`: None
+- `adafactor`: False
+- `group_by_length`: False
+- `length_column_name`: length
+- `ddp_find_unused_parameters`: None
+- `ddp_bucket_cap_mb`: None
+- `ddp_broadcast_buffers`: False
+- `dataloader_pin_memory`: True
+- `dataloader_persistent_workers`: False
+- `skip_memory_metrics`: True
+- `use_legacy_prediction_loop`: False
+- `push_to_hub`: False
+- `resume_from_checkpoint`: None
+- `hub_model_id`: None
+- `hub_strategy`: every_save
+- `hub_private_repo`: None
+- `hub_always_push`: False
+- `gradient_checkpointing`: False
+- `gradient_checkpointing_kwargs`: None
+- `include_inputs_for_metrics`: False
+- `include_for_metrics`: []
+- `eval_do_concat_batches`: True
+- `fp16_backend`: auto
+- `push_to_hub_model_id`: None
+- `push_to_hub_organization`: None
+- `mp_parameters`:
+- `auto_find_batch_size`: False
+- `full_determinism`: False
+- `torchdynamo`: None
+- `ray_scope`: last
+- `ddp_timeout`: 1800
+- `torch_compile`: False
+- `torch_compile_backend`: None
+- `torch_compile_mode`: None
+- `dispatch_batches`: None
+- `split_batches`: None
+- `include_tokens_per_second`: False
+- `include_num_input_tokens_seen`: False
+- `neftune_noise_alpha`: None
+- `optim_target_modules`: None
+- `batch_eval_metrics`: False
+- `eval_on_start`: True
+- `use_liger_kernel`: False
+- `eval_use_gather_object`: False
+- `average_tokens_across_devices`: False
+- `prompts`: None
+- `batch_sampler`: no_duplicates
+- `multi_dataset_batch_sampler`: proportional
+</details>
+### Training Logs
+| Epoch  | Step | Training Loss | Validation Loss | spearman_cosine |
+|:------:|:----:|:-------------:|:---------------:|:---------------:|
+| 0      | 0    | -             | 0.1095          | 0.7843          |
+| 0.1351 | 5    | 0.6784        | 0.0765          | 0.8123          |
+| 0.2703 | 10   | 0.5088        | 0.0533          | 0.8303          |
+| 0.4054 | 15   | 0.4364        | 0.0475          | 0.8339          |
+| 0.5405 | 20   | 0.3456        | 0.0435          | 0.8345          |
+### Framework Versions
+- Python: 3.12.9
+- Sentence Transformers: 3.4.1
+- Transformers: 4.50.0
+- PyTorch: 2.6.0+cpu
+- Accelerate: 1.6.0
+- Datasets: 3.5.0
+- Tokenizers: 0.21.1
+## Citation
+### BibTeX
+#### Sentence Transformers
+```bibtex
+@inproceedings{reimers-2019-sentence-bert,
+    title = "Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks",
+    author = "Reimers, Nils and Gurevych, Iryna",
+    booktitle = "Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing",
+    month = "11",
+    year = "2019",
+    publisher = "Association for Computational Linguistics",
+    url = "https://arxiv.org/abs/1908.10084",
+}
+```
+<!--
+## Glossary
+*Clearly define terms in order to be accessible across audiences.*
+-->
+<!--
+## Model Card Authors
+*Lists the people who create the model card, providing recognition and accountability for the detailed work that goes into its construction.*
+-->
+<!--
+## Model Card Contact
+*Provides a way for people who have updates to the Model Card, suggestions, or questions, to contact the Model Card authors.*
+-->

checkpoint-20/config.json ADDED Viewed

	@@ -0,0 +1,27 @@

+{
+  "architectures": [
+    "XLMRobertaModel"
+  ],
+  "attention_probs_dropout_prob": 0.1,
+  "bos_token_id": 0,
+  "classifier_dropout": null,
+  "eos_token_id": 2,
+  "hidden_act": "gelu",
+  "hidden_dropout_prob": 0.1,
+  "hidden_size": 1024,
+  "initializer_range": 0.02,
+  "intermediate_size": 4096,
+  "layer_norm_eps": 1e-05,
+  "max_position_embeddings": 514,
+  "model_type": "xlm-roberta",
+  "num_attention_heads": 16,
+  "num_hidden_layers": 24,
+  "output_past": true,
+  "pad_token_id": 1,
+  "position_embedding_type": "absolute",
+  "torch_dtype": "float32",
+  "transformers_version": "4.50.0",
+  "type_vocab_size": 1,
+  "use_cache": true,
+  "vocab_size": 250002
+}

checkpoint-20/config_sentence_transformers.json ADDED Viewed

	@@ -0,0 +1,10 @@

+{
+  "__version__": {
+    "sentence_transformers": "3.4.1",
+    "transformers": "4.50.0",
+    "pytorch": "2.6.0+cpu"
+  },
+  "prompts": {},
+  "default_prompt_name": null,
+  "similarity_fn_name": "cosine"
+}

checkpoint-20/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:1d8db9ca7a34661a8ab59ec5df97ad40f2c4e75973337e38d6910ecf9c1a527f
+size 2239607176

checkpoint-20/modules.json ADDED Viewed

	@@ -0,0 +1,20 @@

+[
+  {
+    "idx": 0,
+    "name": "0",
+    "path": "",
+    "type": "sentence_transformers.models.Transformer"
+  },
+  {
+    "idx": 1,
+    "name": "1",
+    "path": "1_Pooling",
+    "type": "sentence_transformers.models.Pooling"
+  },
+  {
+    "idx": 2,
+    "name": "2",
+    "path": "2_Normalize",
+    "type": "sentence_transformers.models.Normalize"
+  }
+]

checkpoint-20/optimizer.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e9a2e5de59454cb420f37bc12a0f0c4f45d5c2ba401324295394063e96b747af
+size 4471044921

checkpoint-20/rng_state.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:f6f5fc60f03b02c0e0a76c701d090afac8e288367825dce0a6ff8fc8463b25ee
+size 13990

checkpoint-20/scheduler.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:66649393eb0b497626eaceb0dc7ef324c049601175c3adb7f3d8bf15359013ec
+size 1064

checkpoint-20/sentence_bert_config.json ADDED Viewed

	@@ -0,0 +1,4 @@

+{
+  "max_seq_length": 512,
+  "do_lower_case": false
+}

checkpoint-20/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,51 @@

+{
+  "bos_token": {
+    "content": "<s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "cls_token": {
+    "content": "<s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "</s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "mask_token": {
+    "content": "<mask>",
+    "lstrip": true,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "<pad>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "sep_token": {
+    "content": "</s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "<unk>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

checkpoint-20/tokenizer.json ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:883b037111086fd4dfebbbc9b7cee11e1517b5e0c0514879478661440f137085
+size 17082987

checkpoint-20/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,55 @@

+{
+  "added_tokens_decoder": {
+    "0": {
+      "content": "<s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "<pad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "</s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "3": {
+      "content": "<unk>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "250001": {
+      "content": "<mask>",
+      "lstrip": true,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "<s>",
+  "clean_up_tokenization_spaces": true,
+  "cls_token": "<s>",
+  "eos_token": "</s>",
+  "extra_special_tokens": {},
+  "mask_token": "<mask>",
+  "model_max_length": 512,
+  "pad_token": "<pad>",
+  "sep_token": "</s>",
+  "tokenizer_class": "XLMRobertaTokenizer",
+  "unk_token": "<unk>"
+}

checkpoint-20/trainer_state.json ADDED Viewed

	@@ -0,0 +1,112 @@

+{
+  "best_global_step": 20,
+  "best_metric": 0.043546345084905624,
+  "best_model_checkpoint": "models/me5-large-retraining\\checkpoint-20",
+  "epoch": 0.5405405405405406,
+  "eval_steps": 5,
+  "global_step": 20,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "epoch": 0,
+      "eval_loss": 0.10950354486703873,
+      "eval_pearson_cosine": 0.7806081242807599,
+      "eval_runtime": 151.6579,
+      "eval_samples_per_second": 6.633,
+      "eval_spearman_cosine": 0.7843279594448466,
+      "eval_steps_per_second": 0.831,
+      "step": 0
+    },
+    {
+      "epoch": 0.13513513513513514,
+      "grad_norm": 9.38223934173584,
+      "learning_rate": 1e-05,
+      "loss": 0.6784,
+      "step": 5
+    },
+    {
+      "epoch": 0.13513513513513514,
+      "eval_loss": 0.07647562772035599,
+      "eval_pearson_cosine": 0.8193313583571735,
+      "eval_runtime": 156.2929,
+      "eval_samples_per_second": 6.437,
+      "eval_spearman_cosine": 0.8122999445241028,
+      "eval_steps_per_second": 0.806,
+      "step": 5
+    },
+    {
+      "epoch": 0.2702702702702703,
+      "grad_norm": 8.665059089660645,
+      "learning_rate": 1e-05,
+      "loss": 0.5088,
+      "step": 10
+    },
+    {
+      "epoch": 0.2702702702702703,
+      "eval_loss": 0.05329965054988861,
+      "eval_pearson_cosine": 0.846437709431274,
+      "eval_runtime": 152.6705,
+      "eval_samples_per_second": 6.589,
+      "eval_spearman_cosine": 0.8303112726354404,
+      "eval_steps_per_second": 0.825,
+      "step": 10
+    },
+    {
+      "epoch": 0.40540540540540543,
+      "grad_norm": 21.96602439880371,
+      "learning_rate": 1e-05,
+      "loss": 0.4364,
+      "step": 15
+    },
+    {
+      "epoch": 0.40540540540540543,
+      "eval_loss": 0.047497160732746124,
+      "eval_pearson_cosine": 0.8497791216636505,
+      "eval_runtime": 150.8647,
+      "eval_samples_per_second": 6.668,
+      "eval_spearman_cosine": 0.8338916292060913,
+      "eval_steps_per_second": 0.835,
+      "step": 15
+    },
+    {
+      "epoch": 0.5405405405405406,
+      "grad_norm": 10.698002815246582,
+      "learning_rate": 1e-05,
+      "loss": 0.3456,
+      "step": 20
+    },
+    {
+      "epoch": 0.5405405405405406,
+      "eval_loss": 0.043546345084905624,
+      "eval_pearson_cosine": 0.8481768671521202,
+      "eval_runtime": 151.8674,
+      "eval_samples_per_second": 6.624,
+      "eval_spearman_cosine": 0.8344803713886917,
+      "eval_steps_per_second": 0.83,
+      "step": 20
+    }
+  ],
+  "logging_steps": 5,
+  "max_steps": 37,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 1,
+  "save_steps": 10,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": true,
+        "should_training_stop": false
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 0.0,
+  "train_batch_size": 8,
+  "trial_name": null,
+  "trial_params": null
+}

checkpoint-20/training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:7b41c6ac4c736e654c2036409a6676535c43e7b515e2e3c97cfc2b06f8c81bf3
+size 5624

eval/similarity_evaluation_results.csv ADDED Viewed

	@@ -0,0 +1,39 @@

+epoch,steps,cosine_pearson,cosine_spearman
+0,0,0.7806081242807599,0.7843279594448466
+0.13513513513513514,5,0.8193313583571735,0.8122999445241028
+0.2702702702702703,10,0.846437709431274,0.8303112726354404
+0.40540540540540543,15,0.8497791216636505,0.8338916292060913
+0.5405405405405406,20,0.8481768671521202,0.8344803713886917
+0.6756756756756757,25,0.8419294236114762,0.8324197823497284
+0.8108108108108109,30,0.8311833878571917,0.8270800450821792
+0.9459459459459459,35,0.8323883964862309,0.8261627064456549
+1.0810810810810811,40,0.8322596480425719,0.8232463910787938
+1.2162162162162162,45,0.8345968763719022,0.8220928743948337
+1.3513513513513513,50,0.8295312877735412,0.8197002635410278
+1.4864864864864864,55,0.8258612961974536,0.8190362223126076
+1.6216216216216215,60,0.8280490163439709,0.8212679517806181
+1.7567567567567568,65,0.8348764332747338,0.8253343633484949
+1.8918918918918919,70,0.8327080612027193,0.8249373111883099
+2.027027027027027,75,0.8292172905350813,0.8226234223032437
+2.1621621621621623,80,0.8348787263877363,0.8262517044196894
+2.2972972972972974,85,0.834739922043313,0.8269978977904043
+2.4324324324324325,90,0.8300953895591171,0.8249920776743955
+2.5675675675675675,95,0.8280036278693699,0.8241911129581995
+2.7027027027027026,100,0.827662426497365,0.8241911153867979
+2.8378378378378377,105,0.8271686930217653,0.8242595734942031
+2.972972972972973,110,0.8242998475213165,0.8243348749833265
+3.108108108108108,115,0.8100599157021782,0.8195085832550942
+3.2432432432432434,120,0.8092739374985023,0.8194743493718913
+3.3783783783783785,125,0.8212737827789367,0.8250605382131625
+3.5135135135135136,130,0.8257663666576973,0.8261900896885362
+3.6486486486486487,135,0.8294106767298322,0.826217475365987
+3.7837837837837838,140,0.8333850683405307,0.8273128026466706
+3.918918918918919,145,0.8334934833047165,0.8261353280714282
+0,0,0.7806081242807599,0.7843279594448466
+0.13513513513513514,5,0.8193313583571735,0.8122999445241028
+0.2702702702702703,10,0.846437709431274,0.8303112726354404
+0.40540540540540543,15,0.8497791216636505,0.8338916292060913
+0.5405405405405406,20,0.8481768671521202,0.8344803713886917
+0.6756756756756757,25,0.8419294236114762,0.8324197823497284
+0.8108108108108109,30,0.8311833878571917,0.8270800450821792
+0.9459459459459459,35,0.8323883964862309,0.8261627064456549

latest/1_Pooling/config.json ADDED Viewed

	@@ -0,0 +1,10 @@

+{
+  "word_embedding_dimension": 1024,
+  "pooling_mode_cls_token": false,
+  "pooling_mode_mean_tokens": true,
+  "pooling_mode_max_tokens": false,
+  "pooling_mode_mean_sqrt_len_tokens": false,
+  "pooling_mode_weightedmean_tokens": false,
+  "pooling_mode_lasttoken": false,
+  "include_prompt": true
+}

latest/README.md ADDED Viewed

	@@ -0,0 +1,404 @@

+---
+tags:
+- sentence-transformers
+- sentence-similarity
+- feature-extraction
+- generated_from_trainer
+- dataset_size:290
+- loss:OnlineContrastiveLoss
+base_model: intfloat/multilingual-e5-large
+widget:
+- source_sentence: Antes se coge al mentiroso que al cojo
+  sentences:
+  - A escudero pobre, taza de plata y cántaro de cobre
+  - En río revuelto, pesca abundante
+  - Se ayuda primero al necesitado que al engañador.
+- source_sentence: Asno de muchos, lobos lo comen
+  sentences:
+  - Sabio entre sabios, amigos lo respetan.
+  - El que mucho madruga más hace que el que Dios ayuda.
+  - Se pilla antes a un mentiroso que a un cojo
+- source_sentence: Al buey por el asta, y al hombre por la palabra
+  sentences:
+  - Si no quieres arroz con leche, toma tres tazas
+  - Al hombre por la palabra, y al buey por el cuerno ata
+  - Ese no es tu amigo, sino alguien que siempre busca estar rodeado de bullicio y
+    actividad.
+- source_sentence: Al médico, confesor y letrado, hablarles claro
+  sentences:
+  - Al médico, confesor y letrado, no le hayas engañado
+  - Más vale a quien Dios ayuda que quien mucho madruga
+  - Al que anda entre la miel, algo se le pega
+- source_sentence: A muertos y a idos, no hay amigos
+  sentences:
+  - Al buen callar llaman santo
+  - A los vivos y presentes, siempre hay amigos.
+  - Al que de prestado se viste, en la calle lo desnudan
+pipeline_tag: sentence-similarity
+library_name: sentence-transformers
+metrics:
+- pearson_cosine
+- spearman_cosine
+model-index:
+- name: SentenceTransformer based on intfloat/multilingual-e5-large
+  results:
+  - task:
+      type: semantic-similarity
+      name: Semantic Similarity
+    dataset:
+      name: Unknown
+      type: unknown
+    metrics:
+    - type: pearson_cosine
+      value: 0.8323883964862309
+      name: Pearson Cosine
+    - type: spearman_cosine
+      value: 0.8261627064456549
+      name: Spearman Cosine
+---
+# SentenceTransformer based on intfloat/multilingual-e5-large
+This is a [sentence-transformers](https://www.SBERT.net) model finetuned from [intfloat/multilingual-e5-large](https://huggingface.co/intfloat/multilingual-e5-large) on the csv dataset. It maps sentences & paragraphs to a 1024-dimensional dense vector space and can be used for semantic textual similarity, semantic search, paraphrase mining, text classification, clustering, and more.
+## Model Details
+### Model Description
+- **Model Type:** Sentence Transformer
+- **Base model:** [intfloat/multilingual-e5-large](https://huggingface.co/intfloat/multilingual-e5-large) <!-- at revision 0dc5580a448e4284468b8909bae50fa925907bc5 -->
+- **Maximum Sequence Length:** 512 tokens
+- **Output Dimensionality:** 1024 dimensions
+- **Similarity Function:** Cosine Similarity
+- **Training Dataset:**
+    - csv
+<!-- - **Language:** Unknown -->
+<!-- - **License:** Unknown -->
+### Model Sources
+- **Documentation:** [Sentence Transformers Documentation](https://sbert.net)
+- **Repository:** [Sentence Transformers on GitHub](https://github.com/UKPLab/sentence-transformers)
+- **Hugging Face:** [Sentence Transformers on Hugging Face](https://huggingface.co/models?library=sentence-transformers)
+### Full Model Architecture
+```
+SentenceTransformer(
+  (0): Transformer({'max_seq_length': 512, 'do_lower_case': False}) with Transformer model: XLMRobertaModel
+  (1): Pooling({'word_embedding_dimension': 1024, 'pooling_mode_cls_token': False, 'pooling_mode_mean_tokens': True, 'pooling_mode_max_tokens': False, 'pooling_mode_mean_sqrt_len_tokens': False, 'pooling_mode_weightedmean_tokens': False, 'pooling_mode_lasttoken': False, 'include_prompt': True})
+  (2): Normalize()
+)
+```
+## Usage
+### Direct Usage (Sentence Transformers)
+First install the Sentence Transformers library:
+```bash
+pip install -U sentence-transformers
+```
+Then you can load this model and run inference.
+```python
+from sentence_transformers import SentenceTransformer
+# Download from the 🤗 Hub
+model = SentenceTransformer("sentence_transformers_model_id")
+# Run inference
+sentences = [
+    'A muertos y a idos, no hay amigos',
+    'A los vivos y presentes, siempre hay amigos.',
+    'Al buen callar llaman santo',
+]
+embeddings = model.encode(sentences)
+print(embeddings.shape)
+# [3, 1024]
+# Get the similarity scores for the embeddings
+similarities = model.similarity(embeddings, embeddings)
+print(similarities.shape)
+# [3, 3]
+```
+<!--
+### Direct Usage (Transformers)
+<details><summary>Click to see the direct usage in Transformers</summary>
+</details>
+-->
+<!--
+### Downstream Usage (Sentence Transformers)
+You can finetune this model on your own dataset.
+<details><summary>Click to expand</summary>
+</details>
+-->
+<!--
+### Out-of-Scope Use
+*List how the model may foreseeably be misused and address what users ought not to do with the model.*
+-->
+## Evaluation
+### Metrics
+#### Semantic Similarity
+* Evaluated with [<code>EmbeddingSimilarityEvaluator</code>](https://sbert.net/docs/package_reference/sentence_transformer/evaluation.html#sentence_transformers.evaluation.EmbeddingSimilarityEvaluator)
+| Metric              | Value      |
+|:--------------------|:-----------|
+| pearson_cosine      | 0.8324     |
+| **spearman_cosine** | **0.8262** |
+<!--
+## Bias, Risks and Limitations
+*What are the known or foreseeable issues stemming from this model? You could also flag here known failure cases or weaknesses of the model.*
+-->
+<!--
+### Recommendations
+*What are recommendations with respect to the foreseeable issues? For example, filtering explicit content.*
+-->
+## Training Details
+### Training Dataset
+#### csv
+* Dataset: csv
+* Size: 290 training samples
+* Columns: <code>sentence1</code>, <code>sentence2</code>, and <code>label</code>
+* Approximate statistics based on the first 290 samples:
+  |         | sentence1                                                                         | sentence2                                                                         | label                                           |
+  |:--------|:----------------------------------------------------------------------------------|:----------------------------------------------------------------------------------|:------------------------------------------------|
+  | type    | string                                                                            | string                                                                            | int                                             |
+  | details | <ul><li>min: 7 tokens</li><li>mean: 11.68 tokens</li><li>max: 22 tokens</li></ul> | <ul><li>min: 7 tokens</li><li>mean: 17.01 tokens</li><li>max: 44 tokens</li></ul> | <ul><li>0: ~50.00%</li><li>1: ~50.00%</li></ul> |
+* Samples:
+  | sentence1                                                   | sentence2                                                                                         | label          |
+  |:------------------------------------------------------------|:--------------------------------------------------------------------------------------------------|:---------------|
+  | <code>Gota a gota, la mar se agota.</code>                  | <code>Con el pasar del tiempo se llega a alcanzar cualquier meta.</code>                          | <code>1</code> |
+  | <code>Dime de qué presumes y te diré de qué careces.</code> | <code>Dime de qué careces y te diré de qué dispones.</code>                                       | <code>0</code> |
+  | <code>Cómo se vive, se muere.</code>                        | <code>De aquella forma que hemos vivido nuestra vida será de la forma en la que moriremos.</code> | <code>1</code> |
+* Loss: [<code>OnlineContrastiveLoss</code>](https://sbert.net/docs/package_reference/sentence_transformer/losses.html#onlinecontrastiveloss)
+### Evaluation Dataset
+#### Unnamed Dataset
+* Size: 1,006 evaluation samples
+* Columns: <code>sentence1</code>, <code>sentence2</code>, and <code>label</code>
+* Approximate statistics based on the first 1000 samples:
+  |         | sentence1                                                                         | sentence2                                                                         | label                                           |
+  |:--------|:----------------------------------------------------------------------------------|:----------------------------------------------------------------------------------|:------------------------------------------------|
+  | type    | string                                                                            | string                                                                            | int                                             |
+  | details | <ul><li>min: 7 tokens</li><li>mean: 12.51 tokens</li><li>max: 25 tokens</li></ul> | <ul><li>min: 6 tokens</li><li>mean: 14.82 tokens</li><li>max: 38 tokens</li></ul> | <ul><li>0: ~49.70%</li><li>1: ~50.30%</li></ul> |
+* Samples:
+  | sentence1                                    | sentence2                                                              | label          |
+  |:---------------------------------------------|:-----------------------------------------------------------------------|:---------------|
+  | <code>¿Adónde irá el buey que no are?</code> | <code>¿A dó irá el buey que no are?</code>                             | <code>1</code> |
+  | <code>¿Adónde irá el buey que no are?</code> | <code>¿Adónde irá el buey que no are ni la  mula que no cargue?</code> | <code>1</code> |
+  | <code>¿Adónde irá el buey que no are?</code> | <code>¿Adónde irá el buey que no are, sino al matadero?</code>         | <code>1</code> |
+* Loss: [<code>OnlineContrastiveLoss</code>](https://sbert.net/docs/package_reference/sentence_transformer/losses.html#onlinecontrastiveloss)
+### Training Hyperparameters
+#### Non-Default Hyperparameters
+- `eval_strategy`: steps
+- `learning_rate`: 1e-05
+- `num_train_epochs`: 1
+- `lr_scheduler_type`: constant
+- `load_best_model_at_end`: True
+- `eval_on_start`: True
+- `batch_sampler`: no_duplicates
+#### All Hyperparameters
+<details><summary>Click to expand</summary>
+- `overwrite_output_dir`: False
+- `do_predict`: False
+- `eval_strategy`: steps
+- `prediction_loss_only`: True
+- `per_device_train_batch_size`: 8
+- `per_device_eval_batch_size`: 8
+- `per_gpu_train_batch_size`: None
+- `per_gpu_eval_batch_size`: None
+- `gradient_accumulation_steps`: 1
+- `eval_accumulation_steps`: None
+- `torch_empty_cache_steps`: None
+- `learning_rate`: 1e-05
+- `weight_decay`: 0.0
+- `adam_beta1`: 0.9
+- `adam_beta2`: 0.999
+- `adam_epsilon`: 1e-08
+- `max_grad_norm`: 1.0
+- `num_train_epochs`: 1
+- `max_steps`: -1
+- `lr_scheduler_type`: constant
+- `lr_scheduler_kwargs`: {}
+- `warmup_ratio`: 0.0
+- `warmup_steps`: 0
+- `log_level`: passive
+- `log_level_replica`: warning
+- `log_on_each_node`: True
+- `logging_nan_inf_filter`: True
+- `save_safetensors`: True
+- `save_on_each_node`: False
+- `save_only_model`: False
+- `restore_callback_states_from_checkpoint`: False
+- `no_cuda`: False
+- `use_cpu`: False
+- `use_mps_device`: False
+- `seed`: 42
+- `data_seed`: None
+- `jit_mode_eval`: False
+- `use_ipex`: False
+- `bf16`: False
+- `fp16`: False
+- `fp16_opt_level`: O1
+- `half_precision_backend`: auto
+- `bf16_full_eval`: False
+- `fp16_full_eval`: False
+- `tf32`: None
+- `local_rank`: 0
+- `ddp_backend`: None
+- `tpu_num_cores`: None
+- `tpu_metrics_debug`: False
+- `debug`: []
+- `dataloader_drop_last`: False
+- `dataloader_num_workers`: 0
+- `dataloader_prefetch_factor`: None
+- `past_index`: -1
+- `disable_tqdm`: False
+- `remove_unused_columns`: True
+- `label_names`: None
+- `load_best_model_at_end`: True
+- `ignore_data_skip`: False
+- `fsdp`: []
+- `fsdp_min_num_params`: 0
+- `fsdp_config`: {'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False}
+- `tp_size`: 0
+- `fsdp_transformer_layer_cls_to_wrap`: None
+- `accelerator_config`: {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}
+- `deepspeed`: None
+- `label_smoothing_factor`: 0.0
+- `optim`: adamw_torch
+- `optim_args`: None
+- `adafactor`: False
+- `group_by_length`: False
+- `length_column_name`: length
+- `ddp_find_unused_parameters`: None
+- `ddp_bucket_cap_mb`: None
+- `ddp_broadcast_buffers`: False
+- `dataloader_pin_memory`: True
+- `dataloader_persistent_workers`: False
+- `skip_memory_metrics`: True
+- `use_legacy_prediction_loop`: False
+- `push_to_hub`: False
+- `resume_from_checkpoint`: None
+- `hub_model_id`: None
+- `hub_strategy`: every_save
+- `hub_private_repo`: None
+- `hub_always_push`: False
+- `gradient_checkpointing`: False
+- `gradient_checkpointing_kwargs`: None
+- `include_inputs_for_metrics`: False
+- `include_for_metrics`: []
+- `eval_do_concat_batches`: True
+- `fp16_backend`: auto
+- `push_to_hub_model_id`: None
+- `push_to_hub_organization`: None
+- `mp_parameters`:
+- `auto_find_batch_size`: False
+- `full_determinism`: False
+- `torchdynamo`: None
+- `ray_scope`: last
+- `ddp_timeout`: 1800
+- `torch_compile`: False
+- `torch_compile_backend`: None
+- `torch_compile_mode`: None
+- `dispatch_batches`: None
+- `split_batches`: None
+- `include_tokens_per_second`: False
+- `include_num_input_tokens_seen`: False
+- `neftune_noise_alpha`: None
+- `optim_target_modules`: None
+- `batch_eval_metrics`: False
+- `eval_on_start`: True
+- `use_liger_kernel`: False
+- `eval_use_gather_object`: False
+- `average_tokens_across_devices`: False
+- `prompts`: None
+- `batch_sampler`: no_duplicates
+- `multi_dataset_batch_sampler`: proportional
+</details>
+### Training Logs
+| Epoch      | Step   | Training Loss | Validation Loss | spearman_cosine |
+|:----------:|:------:|:-------------:|:---------------:|:---------------:|
+| 0          | 0      | -             | 0.1095          | 0.7843          |
+| 0.1351     | 5      | 0.6784        | 0.0765          | 0.8123          |
+| 0.2703     | 10     | 0.5088        | 0.0533          | 0.8303          |
+| 0.4054     | 15     | 0.4364        | 0.0475          | 0.8339          |
+| **0.5405** | **20** | **0.3456**    | **0.0435**      | **0.8345**      |
+| 0.6757     | 25     | 0.1423        | 0.0424          | 0.8324          |
+| 0.8108     | 30     | 0.2852        | 0.0443          | 0.8271          |
+| 0.9459     | 35     | 0.2616        | 0.0514          | 0.8262          |
+* The bold row denotes the saved checkpoint.
+### Framework Versions
+- Python: 3.12.9
+- Sentence Transformers: 3.4.1
+- Transformers: 4.50.0
+- PyTorch: 2.6.0+cpu
+- Accelerate: 1.6.0
+- Datasets: 3.5.0
+- Tokenizers: 0.21.1
+## Citation
+### BibTeX
+#### Sentence Transformers
+```bibtex
+@inproceedings{reimers-2019-sentence-bert,
+    title = "Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks",
+    author = "Reimers, Nils and Gurevych, Iryna",
+    booktitle = "Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing",
+    month = "11",
+    year = "2019",
+    publisher = "Association for Computational Linguistics",
+    url = "https://arxiv.org/abs/1908.10084",
+}
+```
+<!--
+## Glossary
+*Clearly define terms in order to be accessible across audiences.*
+-->
+<!--
+## Model Card Authors
+*Lists the people who create the model card, providing recognition and accountability for the detailed work that goes into its construction.*
+-->
+<!--
+## Model Card Contact
+*Provides a way for people who have updates to the Model Card, suggestions, or questions, to contact the Model Card authors.*
+-->

latest/config.json ADDED Viewed

	@@ -0,0 +1,27 @@

+{
+  "architectures": [
+    "XLMRobertaModel"
+  ],
+  "attention_probs_dropout_prob": 0.1,
+  "bos_token_id": 0,
+  "classifier_dropout": null,
+  "eos_token_id": 2,
+  "hidden_act": "gelu",
+  "hidden_dropout_prob": 0.1,
+  "hidden_size": 1024,
+  "initializer_range": 0.02,
+  "intermediate_size": 4096,
+  "layer_norm_eps": 1e-05,
+  "max_position_embeddings": 514,
+  "model_type": "xlm-roberta",
+  "num_attention_heads": 16,
+  "num_hidden_layers": 24,
+  "output_past": true,
+  "pad_token_id": 1,
+  "position_embedding_type": "absolute",
+  "torch_dtype": "float32",
+  "transformers_version": "4.50.0",
+  "type_vocab_size": 1,
+  "use_cache": true,
+  "vocab_size": 250002
+}

latest/config_sentence_transformers.json ADDED Viewed

	@@ -0,0 +1,10 @@

+{
+  "__version__": {
+    "sentence_transformers": "3.4.1",
+    "transformers": "4.50.0",
+    "pytorch": "2.6.0+cpu"
+  },
+  "prompts": {},
+  "default_prompt_name": null,
+  "similarity_fn_name": "cosine"
+}

latest/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:1d8db9ca7a34661a8ab59ec5df97ad40f2c4e75973337e38d6910ecf9c1a527f
+size 2239607176

latest/modules.json ADDED Viewed

	@@ -0,0 +1,20 @@

+[
+  {
+    "idx": 0,
+    "name": "0",
+    "path": "",
+    "type": "sentence_transformers.models.Transformer"
+  },
+  {
+    "idx": 1,
+    "name": "1",
+    "path": "1_Pooling",
+    "type": "sentence_transformers.models.Pooling"
+  },
+  {
+    "idx": 2,
+    "name": "2",
+    "path": "2_Normalize",
+    "type": "sentence_transformers.models.Normalize"
+  }
+]

latest/sentence_bert_config.json ADDED Viewed

	@@ -0,0 +1,4 @@

+{
+  "max_seq_length": 512,
+  "do_lower_case": false
+}

latest/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,51 @@

+{
+  "bos_token": {
+    "content": "<s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "cls_token": {
+    "content": "<s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "</s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "mask_token": {
+    "content": "<mask>",
+    "lstrip": true,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "<pad>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "sep_token": {
+    "content": "</s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "<unk>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

latest/tokenizer.json ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:883b037111086fd4dfebbbc9b7cee11e1517b5e0c0514879478661440f137085
+size 17082987

latest/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,55 @@

+{
+  "added_tokens_decoder": {
+    "0": {
+      "content": "<s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "<pad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "</s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "3": {
+      "content": "<unk>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "250001": {
+      "content": "<mask>",
+      "lstrip": true,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "<s>",
+  "clean_up_tokenization_spaces": true,
+  "cls_token": "<s>",
+  "eos_token": "</s>",
+  "extra_special_tokens": {},
+  "mask_token": "<mask>",
+  "model_max_length": 512,
+  "pad_token": "<pad>",
+  "sep_token": "</s>",
+  "tokenizer_class": "XLMRobertaTokenizer",
+  "unk_token": "<unk>"
+}

latest/training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:7b41c6ac4c736e654c2036409a6676535c43e7b515e2e3c97cfc2b06f8c81bf3
+size 5624