@inproceedings{kilic-etal-2026-fine,
    title = "Fine-Tuning Small Language Models for Cybersecurity: Data Ordering, Knowledge Distillation, and the Educator Effect",
    author = "Kilic, Ozkan  and
      Soundaramourty, Raja  and
      Chenchaiah, Ramu",
    editor = "Mitkov, Ruslan  and
      Mu{\~n}oz, Rafael  and
      Lloret, Elena  and
      Ranasinghe, Tharindu  and
      Estevanell-Valladares, Ernesto L.  and
      Lamsiyah, Salima  and
      Montoyo, Andr{\'e}s  and
      Ezzini, Saad",
    booktitle = "Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security",
    month = jun,
    year = "2026",
    address = "Alicante, Spain",
    publisher = "Department of Languages and Information Systems, University of Alicante",
    url = "https://aclanthology.org/2026.nlpaics-1.6/",
    pages = "55--63",
    abstract = "Data-sovereignty rules forbid cloud-hosted AI in many high-security environments, leaving compact on-premise models as the only path to AI-assisted cybersecurity. We fine-tune three small open-source models, Gemma 2 2B, Phi-3.5 3.8B, Llama 3.1 8B, on {\textasciitilde}147,600 synthetic cybersecurity QA pairs using QLoRA on V100 GPUs. Under strict MCQ evaluation Gemma 2 gains +9.3 pp, Phi +4.0 pp, and Llama drops {\ensuremath{-}}24.0 pp. We term this the educator effect: models trained on pedagogical data internalize explanatory behavior at the expense of format compliance. Severity appears to scale with capacity, though capacity is confounded with architecture and learning rate. A controlled ablation shows randomized ordering outperforms curriculum, without significance testing on the 75-question exam."
}
<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="kilic-etal-2026-fine">
    <titleInfo>
        <title>Fine-Tuning Small Language Models for Cybersecurity: Data Ordering, Knowledge Distillation, and the Educator Effect</title>
    </titleInfo>
    <name type="personal">
        <namePart type="given">Ozkan</namePart>
        <namePart type="family">Kilic</namePart>
        <role>
            <roleTerm authority="marcrelator" type="text">author</roleTerm>
        </role>
    </name>
    <name type="personal">
        <namePart type="given">Raja</namePart>
        <namePart type="family">Soundaramourty</namePart>
        <role>
            <roleTerm authority="marcrelator" type="text">author</roleTerm>
        </role>
    </name>
    <name type="personal">
        <namePart type="given">Ramu</namePart>
        <namePart type="family">Chenchaiah</namePart>
        <role>
            <roleTerm authority="marcrelator" type="text">author</roleTerm>
        </role>
    </name>
    <originInfo>
        <dateIssued>2026-06</dateIssued>
    </originInfo>
    <typeOfResource>text</typeOfResource>
    <relatedItem type="host">
        <titleInfo>
            <title>Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security</title>
        </titleInfo>
        <name type="personal">
            <namePart type="given">Ruslan</namePart>
            <namePart type="family">Mitkov</namePart>
            <role>
                <roleTerm authority="marcrelator" type="text">editor</roleTerm>
            </role>
        </name>
        <name type="personal">
            <namePart type="given">Rafael</namePart>
            <namePart type="family">Muñoz</namePart>
            <role>
                <roleTerm authority="marcrelator" type="text">editor</roleTerm>
            </role>
        </name>
        <name type="personal">
            <namePart type="given">Elena</namePart>
            <namePart type="family">Lloret</namePart>
            <role>
                <roleTerm authority="marcrelator" type="text">editor</roleTerm>
            </role>
        </name>
        <name type="personal">
            <namePart type="given">Tharindu</namePart>
            <namePart type="family">Ranasinghe</namePart>
            <role>
                <roleTerm authority="marcrelator" type="text">editor</roleTerm>
            </role>
        </name>
        <name type="personal">
            <namePart type="given">Ernesto</namePart>
            <namePart type="given">L</namePart>
            <namePart type="family">Estevanell-Valladares</namePart>
            <role>
                <roleTerm authority="marcrelator" type="text">editor</roleTerm>
            </role>
        </name>
        <name type="personal">
            <namePart type="given">Salima</namePart>
            <namePart type="family">Lamsiyah</namePart>
            <role>
                <roleTerm authority="marcrelator" type="text">editor</roleTerm>
            </role>
        </name>
        <name type="personal">
            <namePart type="given">Andrés</namePart>
            <namePart type="family">Montoyo</namePart>
            <role>
                <roleTerm authority="marcrelator" type="text">editor</roleTerm>
            </role>
        </name>
        <name type="personal">
            <namePart type="given">Saad</namePart>
            <namePart type="family">Ezzini</namePart>
            <role>
                <roleTerm authority="marcrelator" type="text">editor</roleTerm>
            </role>
        </name>
        <originInfo>
            <publisher>Department of Languages and Information Systems, University of Alicante</publisher>
            <place>
                <placeTerm type="text">Alicante, Spain</placeTerm>
            </place>
        </originInfo>
        <genre authority="marcgt">conference publication</genre>
    </relatedItem>
    <abstract>Data-sovereignty rules forbid cloud-hosted AI in many high-security environments, leaving compact on-premise models as the only path to AI-assisted cybersecurity. We fine-tune three small open-source models, Gemma 2 2B, Phi-3.5 3.8B, Llama 3.1 8B, on ~147,600 synthetic cybersecurity QA pairs using QLoRA on V100 GPUs. Under strict MCQ evaluation Gemma 2 gains +9.3 pp, Phi +4.0 pp, and Llama drops \ensuremath-24.0 pp. We term this the educator effect: models trained on pedagogical data internalize explanatory behavior at the expense of format compliance. Severity appears to scale with capacity, though capacity is confounded with architecture and learning rate. A controlled ablation shows randomized ordering outperforms curriculum, without significance testing on the 75-question exam.</abstract>
    <identifier type="citekey">kilic-etal-2026-fine</identifier>
    <location>
        <url>https://aclanthology.org/2026.nlpaics-1.6/</url>
    </location>
    <part>
        <date>2026-06</date>
        <extent unit="page">
            <start>55</start>
            <end>63</end>
        </extent>
    </part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Fine-Tuning Small Language Models for Cybersecurity: Data Ordering, Knowledge Distillation, and the Educator Effect
%A Kilic, Ozkan
%A Soundaramourty, Raja
%A Chenchaiah, Ramu
%Y Mitkov, Ruslan
%Y Muñoz, Rafael
%Y Lloret, Elena
%Y Ranasinghe, Tharindu
%Y Estevanell-Valladares, Ernesto L.
%Y Lamsiyah, Salima
%Y Montoyo, Andrés
%Y Ezzini, Saad
%S Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security
%D 2026
%8 June
%I Department of Languages and Information Systems, University of Alicante
%C Alicante, Spain
%F kilic-etal-2026-fine
%X Data-sovereignty rules forbid cloud-hosted AI in many high-security environments, leaving compact on-premise models as the only path to AI-assisted cybersecurity. We fine-tune three small open-source models, Gemma 2 2B, Phi-3.5 3.8B, Llama 3.1 8B, on ~147,600 synthetic cybersecurity QA pairs using QLoRA on V100 GPUs. Under strict MCQ evaluation Gemma 2 gains +9.3 pp, Phi +4.0 pp, and Llama drops \ensuremath-24.0 pp. We term this the educator effect: models trained on pedagogical data internalize explanatory behavior at the expense of format compliance. Severity appears to scale with capacity, though capacity is confounded with architecture and learning rate. A controlled ablation shows randomized ordering outperforms curriculum, without significance testing on the 75-question exam.
%U https://aclanthology.org/2026.nlpaics-1.6/
%P 55-63
Markdown (Informal)

[Fine-Tuning Small Language Models for Cybersecurity: Data Ordering, Knowledge Distillation, and the Educator Effect](https://aclanthology.org/2026.nlpaics-1.6/) (Kilic et al., NLPAICS 2026)

ACL