mirror of https://github.com/aliasrobotics/cai.git
minor improvements in docs
Signed-off-by: Víctor Mayoral Vilches <v.mayoralv@gmail.com>
This commit is contained in:
parent
f5f3c42eb5
commit
fde4c4aef4
|
|
@ -33,7 +33,7 @@ In rigorous Attack & Defense CTF evaluations, **`alias1` consistently outperform
|
||||||
<th style="text-align:center;"><b>Best Performance in Agent vs Agent A&D</b></th>
|
<th style="text-align:center;"><b>Best Performance in Agent vs Agent A&D</b></th>
|
||||||
</tr>
|
</tr>
|
||||||
<tr>
|
<tr>
|
||||||
<td align="center"><img src="../assets/images/stackplot.png" alt="A&D Performance Stack Plot" /></td>
|
<td align="center"><img src="/assets/images/stackplot.png" alt="A&D Performance Stack Plot" /></td>
|
||||||
</tr>
|
</tr>
|
||||||
</table>
|
</table>
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -18,7 +18,7 @@ Jeopardy-style Capture The Flag (CTF) challenges evaluate AI agents on independe
|
||||||
<th style="text-align:center;"><b>Model Performance in Jeopardy CTFs Base Benchmark</b></th>
|
<th style="text-align:center;"><b>Model Performance in Jeopardy CTFs Base Benchmark</b></th>
|
||||||
</tr>
|
</tr>
|
||||||
<tr>
|
<tr>
|
||||||
<td align="center"><img src="../assets/images/base_1col.png" alt="Base Benchmark Results" /></td>
|
<td align="center"><img src="/assets/images/base_1col.png" alt="Base Benchmark Results" /></td>
|
||||||
</tr>
|
</tr>
|
||||||
</table>
|
</table>
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -37,16 +37,16 @@ AutoPenBench │
|
||||||
<th style="text-align:center;"><b>Model Performance in Jeopardy CTFs</b></th>
|
<th style="text-align:center;"><b>Model Performance in Jeopardy CTFs</b></th>
|
||||||
</tr>
|
</tr>
|
||||||
<tr>
|
<tr>
|
||||||
<td align="center"><img src="../assets/images/stackplot.png" alt="A&D Performance" width="100%" /></td>
|
<td align="center"><img src="/assets/images/stackplot.png" alt="A&D Performance" width="100%" /></td>
|
||||||
<td align="center"><img src="../assets/images/base_1col.png" alt="Jeopardy CTF Performance" width="100%" /></td>
|
<td align="center"><img src="/assets/images/base_1col.png" alt="Jeopardy CTF Performance" width="100%" /></td>
|
||||||
</tr>
|
</tr>
|
||||||
<tr>
|
<tr>
|
||||||
<th style="text-align:center;"><b>Model Performance in Privacy Benchmark</b></th>
|
<th style="text-align:center;"><b>Model Performance in Privacy Benchmark</b></th>
|
||||||
<th style="text-align:center;"><b>Overall Model Performance</b></th>
|
<th style="text-align:center;"><b>Overall Model Performance</b></th>
|
||||||
</tr>
|
</tr>
|
||||||
<tr>
|
<tr>
|
||||||
<td align="center"><img src="../assets/images/cyberpii_benchmark.png" alt="Privacy Benchmark" width="100%" /></td>
|
<td align="center"><img src="/assets/images/cyberpii_benchmark.png" alt="Privacy Benchmark" width="100%" /></td>
|
||||||
<td align="center"><img src="../assets/images/caibench_spider.png" alt="Overall Performance" width="100%" /></td>
|
<td align="center"><img src="/assets/images/caibench_spider.png" alt="Overall Performance" width="100%" /></td>
|
||||||
</tr>
|
</tr>
|
||||||
</table>
|
</table>
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -13,7 +13,7 @@ Privacy benchmarks assess AI models' ability to handle sensitive information app
|
||||||
<th style="text-align:center;"><b>Model Performance in CyberPII Privacy Benchmark</b></th>
|
<th style="text-align:center;"><b>Model Performance in CyberPII Privacy Benchmark</b></th>
|
||||||
</tr>
|
</tr>
|
||||||
<tr>
|
<tr>
|
||||||
<td align="center"><img src="../assets/images/cyberpii_benchmark.png" alt="CyberPII Benchmark Results" /></td>
|
<td align="center"><img src="/assets/images/cyberpii_benchmark.png" alt="CyberPII Benchmark Results" /></td>
|
||||||
</tr>
|
</tr>
|
||||||
</table>
|
</table>
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -168,27 +168,55 @@ Cybersecurity AI is a critical field, yet many groups are misguidedly pursuing i
|
||||||
If you want to cite our work, please use the following:
|
If you want to cite our work, please use the following:
|
||||||
|
|
||||||
```bibtex
|
```bibtex
|
||||||
@misc{mayoralvilches2025caiopenbugbountyready,
|
@article{mayoral2025cai,
|
||||||
title={CAI: An Open, Bug Bounty-Ready Cybersecurity AI},
|
title={CAI: An Open, Bug Bounty-Ready Cybersecurity AI},
|
||||||
author={Víctor Mayoral-Vilches and Luis Javier Navarrete-Lozano and María Sanz-Gómez and Lidia Salas Espejo and Martiño Crespo-Álvarez and Francisco Oca-Gonzalez and Francesco Balassone and Alfonso Glera-Picón and Unai Ayucar-Carbajo and Jon Ander Ruiz-Alcalde and Stefan Rass and Martin Pinzger and Endika Gil-Uriarte},
|
author={Mayoral-Vilches, V{\'\i}ctor and Navarrete-Lozano, Luis Javier and Sanz-G{\'o}mez, Mar{\'\i}a and Espejo, Lidia Salas and Crespo-{\'A}lvarez, Marti{\~n}o and Oca-Gonzalez, Francisco and Balassone, Francesco and Glera-Pic{\'o}n, Alfonso and Ayucar-Carbajo, Unai and Ruiz-Alcalde, Jon Ander and Rass, Stefan and Pinzger, Martin and Gil-Uriarte, Endika},
|
||||||
year={2025},
|
journal={arXiv preprint arXiv:2504.06017},
|
||||||
eprint={2504.06017},
|
year={2025}
|
||||||
archivePrefix={arXiv},
|
|
||||||
primaryClass={cs.CR},
|
|
||||||
url={https://arxiv.org/abs/2504.06017},
|
|
||||||
}
|
}
|
||||||
```
|
|
||||||
|
|
||||||
```bibtex
|
@article{mayoral2025automation,
|
||||||
@misc{mayoralvilches2025cybersecurityaidangerousgap,
|
title={Cybersecurity AI: The Dangerous Gap Between Automation and Autonomy},
|
||||||
title={Cybersecurity AI: The Dangerous Gap Between Automation and Autonomy},
|
author={Mayoral-Vilches, V{\'\i}ctor},
|
||||||
author={Víctor Mayoral-Vilches},
|
journal={arXiv preprint arXiv:2506.23592},
|
||||||
year={2025},
|
year={2025}
|
||||||
eprint={2506.23592},
|
|
||||||
archivePrefix={arXiv},
|
|
||||||
primaryClass={cs.CR},
|
|
||||||
url={https://arxiv.org/abs/2506.23592},
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
@article{mayoral2025fluency,
|
||||||
|
title={CAI Fluency: A Framework for Cybersecurity AI Fluency},
|
||||||
|
author={Mayoral-Vilches, V{\'\i}ctor and Wachter, Jasmin and Chavez, Crist{\'o}bal RJ and Schachner, Cathrin and Navarrete-Lozano, Luis Javier and Sanz-G{\'o}mez, Mar{\'\i}a},
|
||||||
|
journal={arXiv preprint arXiv:2508.13588},
|
||||||
|
year={2025}
|
||||||
|
}
|
||||||
|
|
||||||
|
@article{mayoral2025hacking,
|
||||||
|
title={Cybersecurity AI: Hacking the AI Hackers via Prompt Injection},
|
||||||
|
author={Mayoral-Vilches, V{\'\i}ctor and Rynning, Per Mannermaa},
|
||||||
|
journal={arXiv preprint arXiv:2508.21669},
|
||||||
|
year={2025}
|
||||||
|
}
|
||||||
|
|
||||||
|
@article{mayoral2025humanoid,
|
||||||
|
title={Cybersecurity AI: Humanoid Robots as Attack Vectors},
|
||||||
|
author={Mayoral-Vilches, V{\'\i}ctor},
|
||||||
|
journal={arXiv preprint arXiv:2509.14139},
|
||||||
|
year={2025}
|
||||||
|
}
|
||||||
|
|
||||||
|
@article{balassone2025evaluation,
|
||||||
|
title={Cybersecurity AI: Evaluating Agentic Cybersecurity in Attack/Defense CTFs},
|
||||||
|
author={Balassone, Francesco and Mayoral-Vilches, V{\'\i}ctor and Rass, Stefan and Pinzger, Martin and Perrone, Gaetano and Romano, Simon Pietro and Schartner, Peter},
|
||||||
|
journal={arXiv preprint arXiv:2510.17521},
|
||||||
|
year={2025}
|
||||||
|
}
|
||||||
|
|
||||||
|
@article{mayoral2025caibench,
|
||||||
|
title={CAIBench: A Meta-Benchmark for Evaluating Cybersecurity AI Agents},
|
||||||
|
author={Mayoral-Vilches, V{\'\i}ctor and Balassone, Francesco and Navarrete-Lozano, Luis Javier and Sanz-G{\'o}mez, Mar{\'\i}a and Crespo-{\'A}lvarez, Marti{\~n}o and Rass, Stefan and Pinzger, Martin},
|
||||||
|
journal={arXiv preprint arXiv:2510.24317},
|
||||||
|
year={2025}
|
||||||
|
}
|
||||||
|
|
||||||
```
|
```
|
||||||
|
|
||||||
## Acknowledgements
|
## Acknowledgements
|
||||||
|
|
|
||||||
Loading…
Reference in New Issue