Zagreus ITALIC SFT

Fine-tuned from mii-llm/zagreus-0.4B-ita for the Italian Post-Training Challenge.

Training Data

External MCQ data only:

  • sapienzanlp/mmlu_italian splits fewshot and validation
  • s-conia/mmlu_italian split validation
  • li-lab/MMLU-ProX-Lite config it, split validation

ITALIC benchmark answers were not used for training.

ITALIC Evaluation

{
  "base_model": "mii-llm/zagreus-0.4B-ita",
  "output_repo": "FinancialSupport/zagreus-0.4b-ita-italic-sft-codex",
  "train_mode": "dpo",
  "target_style": "letter",
  "baseline_quick": null,
  "variants": [
    {
      "variant": {
        "name": "dpo_beta010_lr1e-6_ep1",
        "learning_rate": 1e-06,
        "epochs": 1.0,
        "beta": 0.1,
        "nll_weight": 0.05,
        "batch_size": 2,
        "gradient_accumulation_steps": 16
      },
      "model_dir": "/tmp/zagreus_italic_work/runs/dpo_beta010_lr1e-6_ep1",
      "quick_eval": null
    }
  ],
  "best_variant": {
    "name": "dpo_beta010_lr1e-6_ep1",
    "learning_rate": 1e-06,
    "epochs": 1.0,
    "beta": 0.1,
    "nll_weight": 0.05,
    "batch_size": 2,
    "gradient_accumulation_steps": 16
  },
  "full_eval": {
    "label": "dpo_beta010_lr1e-6_ep1_full",
    "limit": "null",
    "result_path": "/tmp/zagreus_italic_work/eval_results/dpo_beta010_lr1e-6_ep1_full/custom_openai_eval_model_italic_fast_few_shot.json",
    "metrics": {
      "accuracy": 0.292,
      "accuracy:std": 0.45468230667137244
    },
    "invalid_answers": 0,
    "prediction_counts": {
      "B": 4172,
      "A": 5821,
      "C": 6,
      "D": 1
    },
    "macro_accuracy": {
      "culture and commonsense": {
        "accuracy": 0.2918203151190077,
        "n": 5966
      },
      "language capability": {
        "accuracy": 0.2922657411998017,
        "n": 4034
      }
    },
    "category_accuracy": {
      "art_history": {
        "accuracy": 0.2836734693877551,
        "n": 980
      },
      "civic_education": {
        "accuracy": 0.31654676258992803,
        "n": 973
      },
      "current_events": {
        "accuracy": 0.32608695652173914,
        "n": 92
      },
      "geography": {
        "accuracy": 0.2522982635342186,
        "n": 979
      },
      "history": {
        "accuracy": 0.2607361963190184,
        "n": 978
      },
      "lexicon": {
        "accuracy": 0.27987742594484166,
        "n": 979
      },
      "literature": {
        "accuracy": 0.2845528455284553,
        "n": 984
      },
      "morphology": {
        "accuracy": 0.2785714285714286,
        "n": 140
      },
      "orthography": {
        "accuracy": 0.282183316168898,
        "n": 971
      },
      "synonyms_and_antonyms": {
        "accuracy": 0.32852729145211124,
        "n": 971
      },
      "syntax": {
        "accuracy": 0.2805755395683453,
        "n": 973
      },
      "tourism": {
        "accuracy": 0.35,
        "n": 980
      }
    }
  },
  "train_records": 13074,
  "tokenized_train_records": 29976,
  "italic_revision": "92df420ff686babeea54e217b9f90f8471374916",
  "max_length": 1536,
  "use_eval_few_shots": true,
  "augment_rotations": 5,
  "four_option_only": false,
  "include_extra_mcq": false,
  "global_mmlu_limit": 6000
}
Downloads last month
20
Safetensors
Model size
0.4B params
Tensor type
BF16
·
Inference Providers NEW
This model isn't deployed by any Inference Provider. 🙋 Ask for provider support

Model tree for FinancialSupport/zagreus-0.4b-ita-italic-sft-codex

Finetuned
(13)
this model