diff --git a/README.md b/README.md
index 9012254e53..a6f51e1d4b 100644
--- a/README.md
+++ b/README.md
@@ -232,6 +232,7 @@ Now you can try to find the secrets by means of solving the challenge offered at
- [localhost:8080/challenge/challenge-71](http://localhost:8080/challenge/challenge-71)
- [localhost:8080/challenge/challenge-72](http://localhost:8080/challenge/challenge-72)
- [localhost:8080/challenge/challenge-73](http://localhost:8080/challenge/challenge-73)
+- [localhost:8080/challenge/challenge-76](http://localhost:8080/challenge/challenge-76)
Note that these challenges are still very basic, and so are their explanations. Feel free to file a PR to make them look
diff --git a/pom.xml b/pom.xml
index 619512873e..fb695ab989 100644
--- a/pom.xml
+++ b/pom.xml
@@ -69,6 +69,7 @@
26
3.7.1
10.1.2.0
+ 3.0.6
1.18.48
3.16.0
3.11.0
@@ -329,6 +330,12 @@
test
+
+ io.github.jbellis
+ jvector
+ ${jvector.version}
+
+
org.cyclonedx
cyclonedx-core-java
diff --git a/src/main/java/org/owasp/wrongsecrets/challenges/docker/Challenge76.java b/src/main/java/org/owasp/wrongsecrets/challenges/docker/Challenge76.java
new file mode 100644
index 0000000000..5e1c163941
--- /dev/null
+++ b/src/main/java/org/owasp/wrongsecrets/challenges/docker/Challenge76.java
@@ -0,0 +1,195 @@
+package org.owasp.wrongsecrets.challenges.docker;
+
+import com.google.common.base.Supplier;
+import com.google.common.base.Suppliers;
+import io.github.jbellis.jvector.graph.GraphIndex;
+import io.github.jbellis.jvector.graph.GraphIndexBuilder;
+import io.github.jbellis.jvector.graph.GraphSearcher;
+import io.github.jbellis.jvector.graph.ListRandomAccessVectorValues;
+import io.github.jbellis.jvector.graph.RandomAccessVectorValues;
+import io.github.jbellis.jvector.graph.similarity.BuildScoreProvider;
+import io.github.jbellis.jvector.graph.similarity.SearchScoreProvider;
+import io.github.jbellis.jvector.util.Bits;
+import io.github.jbellis.jvector.vector.VectorSimilarityFunction;
+import io.github.jbellis.jvector.vector.VectorizationProvider;
+import io.github.jbellis.jvector.vector.types.VectorFloat;
+import io.github.jbellis.jvector.vector.types.VectorTypeSupport;
+import java.io.IOException;
+import java.io.UncheckedIOException;
+import java.util.ArrayList;
+import java.util.LinkedHashMap;
+import java.util.List;
+import java.util.Locale;
+import java.util.Map;
+import lombok.extern.slf4j.Slf4j;
+import org.owasp.wrongsecrets.challenges.FixedAnswerChallenge;
+import org.springframework.stereotype.Component;
+
+/**
+ * Challenge about a retrieval augmented generation (RAG) vector store that indexed development
+ * documents containing a synthetic secret. The chunks are indexed with JVector, a real in-memory
+ * vector search library, and are exposed through an unauthenticated search endpoint, exactly like a
+ * misconfigured vector database that is reachable without authentication.
+ */
+@Slf4j
+@Component
+public class Challenge76 extends FixedAnswerChallenge {
+
+ static final String DEVELOPMENT_SECRET = "vector-store-dev-secret-e5c91a7b";
+
+ private static final List INDEXED_CHUNKS =
+ List.of(
+ new IndexedChunk(
+ "chunk-0",
+ "Onboarding guide: the staging environment is reset every night at 02:00 UTC."),
+ new IndexedChunk(
+ "chunk-1",
+ "Runbook: restart the ingestion worker with `kubectl rollout restart"
+ + " deployment/ingestion`."),
+ new IndexedChunk(
+ "chunk-2",
+ "Draft API note: the development vector store uses the credential "
+ + DEVELOPMENT_SECRET
+ + " for the nightly embedding job. Rotate before production."),
+ new IndexedChunk(
+ "chunk-3", "Meeting notes: evaluation set v3 improved recall from 0.71 to 0.78."));
+
+ /**
+ * JVector's COSINE similarity is mapped to {@code (1 + cosine) / 2}, so 0.5f is the score of two
+ * vectors without any term in common. Only chunks that share at least one term with the query
+ * score above it and are therefore considered relevant.
+ */
+ private static final float MIN_RELEVANCE_SCORE = 0.5f;
+
+ private static final int GRAPH_MAX_DEGREE = 16;
+ private static final int GRAPH_CONSTRUCTION_DEPTH = 100;
+ private static final float GRAPH_NEIGHBOR_OVERFLOW = 1.2f;
+ private static final float GRAPH_ALPHA = 1.2f;
+
+ private final Supplier vectorStore = Suppliers.memoize(VectorStore::new);
+
+ /** A single text chunk as it is stored in the vector store. */
+ public record IndexedChunk(String id, String text) {}
+
+ @Override
+ public String getAnswer() {
+ return DEVELOPMENT_SECRET;
+ }
+
+ /**
+ * Similarity search against the in-memory vector store: the query is embedded and an approximate
+ * nearest neighbour search runs over the indexed chunks. Every chunk the index scores above the
+ * relevance threshold is returned, including the chunk that leaks the development secret.
+ *
+ * @param query the search query, or null/blank to return every indexed chunk
+ * @return the relevant indexed chunks, most similar first
+ */
+ public List search(String query) {
+ if (query == null || query.isBlank()) {
+ return vectorStore.get().allChunks();
+ }
+ return vectorStore.get().similaritySearch(query);
+ }
+
+ /**
+ * An in-memory RAG vector store backed by JVector: every chunk is embedded into a bag-of-words
+ * vector over the vocabulary of the indexed documents and indexed in an in-memory HNSW graph.
+ * Nothing leaves the application: the index is built once at startup and queried locally.
+ */
+ private static final class VectorStore {
+
+ private final Map vocabulary = new LinkedHashMap<>();
+ private final RandomAccessVectorValues vectors;
+ private final GraphIndex index;
+
+ private VectorStore() {
+ for (IndexedChunk chunk : INDEXED_CHUNKS) {
+ for (var token : tokenize(chunk.text())) {
+ vocabulary.putIfAbsent(token, vocabulary.size());
+ }
+ }
+ List> embeddings = new ArrayList<>();
+ for (IndexedChunk chunk : INDEXED_CHUNKS) {
+ embeddings.add(vts().createFloatVector(embeddingFor(chunk.text())));
+ }
+ this.vectors = new ListRandomAccessVectorValues(embeddings, vocabulary.size());
+ var buildScoreProvider =
+ BuildScoreProvider.randomAccessScoreProvider(vectors, VectorSimilarityFunction.COSINE);
+ try (var builder =
+ new GraphIndexBuilder(
+ buildScoreProvider,
+ vocabulary.size(),
+ GRAPH_MAX_DEGREE,
+ GRAPH_CONSTRUCTION_DEPTH,
+ GRAPH_NEIGHBOR_OVERFLOW,
+ GRAPH_ALPHA)) {
+ this.index = builder.build(vectors);
+ } catch (IOException e) {
+ throw new UncheckedIOException("Could not build the vector index for challenge 76", e);
+ }
+ log.info(
+ "Indexed {} chunks of challenge 76 into an in-memory vector store with {} dimensions",
+ INDEXED_CHUNKS.size(),
+ vocabulary.size());
+ }
+
+ private List allChunks() {
+ return INDEXED_CHUNKS;
+ }
+
+ private List similaritySearch(String query) {
+ var queryVector = vts().createFloatVector(embeddingFor(query));
+ if (isEmpty(queryVector)) {
+ return List.of();
+ }
+ try (var searcher = new GraphSearcher(index)) {
+ var searchScoreProvider =
+ SearchScoreProvider.exact(queryVector, VectorSimilarityFunction.COSINE, vectors);
+ var result = searcher.search(searchScoreProvider, INDEXED_CHUNKS.size(), Bits.ALL);
+ List matches = new ArrayList<>();
+ for (var nodeScore : result.getNodes()) {
+ if (nodeScore.score > MIN_RELEVANCE_SCORE) {
+ matches.add(INDEXED_CHUNKS.get(nodeScore.node));
+ }
+ }
+ return matches;
+ } catch (IOException e) {
+ throw new UncheckedIOException("Could not search the vector index of challenge 76", e);
+ }
+ }
+
+ private float[] embeddingFor(String text) {
+ float[] embedding = new float[vocabulary.size()];
+ for (var token : tokenize(text)) {
+ var dimension = vocabulary.get(token);
+ if (dimension != null) {
+ embedding[dimension] += 1f;
+ }
+ }
+ return embedding;
+ }
+
+ private static boolean isEmpty(VectorFloat> vector) {
+ for (int i = 0; i < vector.length(); i++) {
+ if (vector.get(i) != 0f) {
+ return false;
+ }
+ }
+ return true;
+ }
+
+ private static List tokenize(String text) {
+ List tokens = new ArrayList<>();
+ for (var token : text.toLowerCase(Locale.ROOT).split("[^a-z0-9]+")) {
+ if (!token.isEmpty()) {
+ tokens.add(token);
+ }
+ }
+ return tokens;
+ }
+
+ private static VectorTypeSupport vts() {
+ return VectorizationProvider.getInstance().getVectorTypeSupport();
+ }
+ }
+}
diff --git a/src/main/java/org/owasp/wrongsecrets/challenges/docker/Challenge76Controller.java b/src/main/java/org/owasp/wrongsecrets/challenges/docker/Challenge76Controller.java
new file mode 100644
index 0000000000..f6711dbcfc
--- /dev/null
+++ b/src/main/java/org/owasp/wrongsecrets/challenges/docker/Challenge76Controller.java
@@ -0,0 +1,29 @@
+package org.owasp.wrongsecrets.challenges.docker;
+
+import java.util.List;
+import lombok.RequiredArgsConstructor;
+import lombok.extern.slf4j.Slf4j;
+import org.springframework.web.bind.annotation.GetMapping;
+import org.springframework.web.bind.annotation.RequestParam;
+import org.springframework.web.bind.annotation.RestController;
+
+/** REST controller for Challenge 76 exposing the in-memory RAG vector store search. */
+@Slf4j
+@RestController
+@RequiredArgsConstructor
+public class Challenge76Controller {
+
+ private final Challenge76 challenge;
+
+ /**
+ * Unauthenticated search endpoint of the vector store. It returns the indexed text chunks that
+ * the store scores as relevant to the query, leaking the development secret that was indexed
+ * along with the other documents.
+ */
+ @GetMapping("/rag/search")
+ public List search(
+ @RequestParam(value = "q", required = false) String query) {
+ log.info("Searching the in-memory vector store for Challenge 76...");
+ return challenge.search(query);
+ }
+}
diff --git a/src/main/resources/explanations/challenge76.adoc b/src/main/resources/explanations/challenge76.adoc
new file mode 100644
index 0000000000..b23c0f0f99
--- /dev/null
+++ b/src/main/resources/explanations/challenge76.adoc
@@ -0,0 +1,29 @@
+=== Challenge 76: Find the Secret Indexed in the RAG Vector Store
+
+Retrieval augmented generation (RAG) pipelines answer questions by first looking
+up relevant passages in a vector store and then feeding those passages to a large
+language model. To build that store, documents are split into chunks, turned into
+embeddings and indexed. The chunks themselves stay readable: a search endpoint
+returns the original text of every chunk it considers relevant.
+
+That is exactly what went wrong in this challenge. A development team indexed its
+runbooks, meeting notes and draft API documentation into a vector store that is
+reachable from the application without any authentication:
+
+`GET /rag/search`
+
+The endpoint takes an optional query parameter `q` and returns the indexed text
+chunks that the vector store scores as relevant to it. Without a query it returns
+every indexed chunk. One of the
+chunks was indexed while it still contained a draft note with the credential of
+the nightly embedding job.
+
+Browse the indexed chunks through the endpoint and submit the credential you find
+in them.
+
+[NOTE]
+====
+The endpoint and every credential in this challenge are fictional. Nothing is
+sent over the network: the application embeds the chunks and searches its own
+in-memory vector index at startup.
+====
diff --git a/src/main/resources/explanations/challenge76_de.adoc b/src/main/resources/explanations/challenge76_de.adoc
new file mode 100644
index 0000000000..ff2266cacf
--- /dev/null
+++ b/src/main/resources/explanations/challenge76_de.adoc
@@ -0,0 +1,32 @@
+=== Challenge 76: Finde das Geheimnis im RAG-Vektorstore
+
+Retrieval-Augmented-Generation-(RAG)-Pipelines beantworten Fragen, indem sie zuerst
+relevante Passagen in einem Vektorstore nachschlagen und diese Passagen dann an ein
+großes Sprachmodell übergeben. Um diesen Store aufzubauen, werden Dokumente in
+Chunks zerlegt, in Embeddings umgewandelt und indiziert. Die Chunks selbst bleiben
+lesbar: Ein Suchendpoint gibt den Originaltext jedes Chunks zurück, der als relevant
+gilt.
+
+Genau das ist in dieser Challenge schiefgegangen. Ein Entwicklungsteam hat seine
+Runbooks, Meeting-Notizen und API-Entwurfsdokumentation in einem Vektorstore
+indiziert, der aus der Anwendung heraus ohne jegliche Authentifizierung erreichbar
+ist:
+
+`GET /rag/search`
+
+Der Endpoint akzeptiert einen optionalen Query-Parameter `q` und gibt die
+indizierten Textchunks zurück, die der Store dafür als relevant bewertet. Ohne
+Query gibt er jeden indizierten Chunk zurück. Einer der Chunks wurde indiziert,
+als er noch eine Entwurfsnotiz mit dem Zugangscredential des nächtlichen
+Embedding-Jobs enthielt.
+
+Durchsuche die indizierten Chunks über den Endpoint und reiche das darin gefundene
+Zugangscredential ein.
+
+[NOTE]
+====
+Der Endpoint und jedes Zugangscredential in dieser Challenge sind fiktiv. Nichts
+wird über das Netzwerk gesendet: Die Anwendung wandelt die Chunks beim Start in
+Embeddings um, indiziert sie im eigenen Vektorindex im Speicher und durchsucht
+ausschließlich diesen.
+====
diff --git a/src/main/resources/explanations/challenge76_es.adoc b/src/main/resources/explanations/challenge76_es.adoc
new file mode 100644
index 0000000000..4ae88c88f1
--- /dev/null
+++ b/src/main/resources/explanations/challenge76_es.adoc
@@ -0,0 +1,30 @@
+=== Desafío 76: Encuentra el secreto indexado en el almacén vectorial RAG
+
+Los pipelines de generación aumentada por recuperación (RAG) responden preguntas
+buscando primero los pasajes relevantes en un almacén vectorial y alimentando
+después esos pasajes a un gran modelo de lenguaje. Para construir ese almacén, los
+documentos se dividen en fragmentos, se convierten en embeddings y se indexan. Los
+fragmentos en sí siguen siendo legibles: un endpoint de búsqueda devuelve el texto
+original de cada fragmento que considera relevante.
+
+Exactamente eso es lo que salió mal en este desafío. Un equipo de desarrollo indexó
+sus runbooks, notas de reuniones y documentación de API en borrador en un almacén
+vectorial que es accesible desde la aplicación sin ninguna autenticación:
+
+`GET /rag/search`
+
+El endpoint acepta un parámetro de consulta opcional `q` y devuelve los fragmentos
+de texto indexados que el almacén considera relevantes para la consulta. Sin una
+consulta devuelve todos los fragmentos indexados. Uno de los fragmentos fue
+indexado mientras aún contenía una nota en borrador con la credencial del trabajo
+nocturno de embeddings.
+
+Recorre los fragmentos indexados a través del endpoint y envía la credencial que
+encuentres en ellos.
+
+[NOTE]
+====
+El endpoint y todas las credenciales de este desafío son ficticios. No se envía
+nada por la red: la aplicación convierte los fragmentos en embeddings y los indexa
+en su propio índice vectorial en memoria al iniciarse, y solo busca en él.
+====
diff --git a/src/main/resources/explanations/challenge76_fr.adoc b/src/main/resources/explanations/challenge76_fr.adoc
new file mode 100644
index 0000000000..4d4db5487e
--- /dev/null
+++ b/src/main/resources/explanations/challenge76_fr.adoc
@@ -0,0 +1,32 @@
+=== Challenge 76 : Trouvez le secret indexé dans le magasin vectoriel RAG
+
+Les pipelines de génération augmentée par récupération (RAG) répondent aux
+questions en recherchant d'abord les passages pertinents dans un magasin vectoriel,
+puis en alimentant ces passages avec un grand modèle de langage. Pour construire ce
+magasin, les documents sont découpés en fragments, convertis en embeddings et
+indexés. Les fragments eux-mêmes restent lisibles : un point d'accès de recherche
+renvoie le texte original de chaque fragment jugé pertinent.
+
+C'est exactement ce qui s'est mal passé dans ce challenge. Une équipe de
+développement a indexé ses runbooks, comptes rendus de réunion et documentation API
+de brouillon dans un magasin vectoriel accessible depuis l'application sans aucune
+authentification :
+
+`GET /rag/search`
+
+Le point d'accès accepte un paramètre de requête optionnel `q` et renvoie les
+fragments de texte indexés que le magasin juge pertinents par rapport à elle. Sans
+requête, il renvoie tous les fragments indexés. L'un des fragments a été indexé
+alors qu'il contenait encore une note de brouillon avec l'identifiant du travail
+d'embedding nocturne.
+
+Parcourez les fragments indexés via le point d'accès et soumettez l'identifiant que
+vous y trouvez.
+
+[NOTE]
+====
+Le point d'accès et tous les identifiants de ce challenge sont fictifs. Rien n'est
+envoyé sur le réseau : l'application convertit les fragments en embeddings et les
+indexe dans son propre index vectoriel en mémoire au démarrage, puis ne fait que
+chercher dans celui-ci.
+====
diff --git a/src/main/resources/explanations/challenge76_hint.adoc b/src/main/resources/explanations/challenge76_hint.adoc
new file mode 100644
index 0000000000..55e1371376
--- /dev/null
+++ b/src/main/resources/explanations/challenge76_hint.adoc
@@ -0,0 +1,14 @@
+You can solve this challenge using the following steps:
+
+1. Open the unauthenticated search endpoint of the in-memory vector store:
+- `http://localhost:8080/rag/search`
+- It returns all four indexed text chunks as JSON.
+
+2. Read the chunks and look for the one that mentions a credential:
+- The chunk with the "Draft API note" contains the credential of the nightly
+ embedding job.
+
+3. Submit that credential (everything after "uses the credential ") as the answer.
+
+You can also narrow the search, for example with `q=credential`, but the secret is
+only exposed because the chunk was indexed before the draft note was scrubbed.
diff --git a/src/main/resources/explanations/challenge76_hint_de.adoc b/src/main/resources/explanations/challenge76_hint_de.adoc
new file mode 100644
index 0000000000..476c30975e
--- /dev/null
+++ b/src/main/resources/explanations/challenge76_hint_de.adoc
@@ -0,0 +1,16 @@
+Du kannst diese Challenge mit den folgenden Schritten lösen:
+
+1. Öffne den unauthentifizierten Suchendpoint des Vektorstores im Speicher:
+- `http://localhost:8080/rag/search`
+- Er gibt alle vier indizierten Textchunks als JSON zurück.
+
+2. Lies die Chunks und suche nach demjenigen, der ein Zugangscredential erwähnt:
+- Der Chunk mit der "Draft API note" enthält das Zugangscredential des nächtlichen
+ Embedding-Jobs.
+
+3. Reiche dieses Zugangscredential als Antwort ein (alles nach
+ "uses the credential ").
+
+Du kannst die Suche auch eingrenzen, zum Beispiel mit `q=credential`, aber das
+Geheimnis liegt nur offen, weil der Chunk indiziert wurde, bevor die Entwurfsnotiz
+bereinigt wurde.
diff --git a/src/main/resources/explanations/challenge76_hint_es.adoc b/src/main/resources/explanations/challenge76_hint_es.adoc
new file mode 100644
index 0000000000..7fc0361d55
--- /dev/null
+++ b/src/main/resources/explanations/challenge76_hint_es.adoc
@@ -0,0 +1,16 @@
+Puedes resolver este desafío con los siguientes pasos:
+
+1. Abre el endpoint de búsqueda sin autenticación del almacén vectorial en memoria:
+- `http://localhost:8080/rag/search`
+- Devuelve los cuatro fragmentos de texto indexados como JSON.
+
+2. Lee los fragmentos y busca el que menciona una credencial:
+- El fragmento con la "Draft API note" contiene la credencial del trabajo nocturno
+ de embeddings.
+
+3. Envía esa credencial como respuesta (todo lo que hay después de
+ "uses the credential ").
+
+También puedes acotar la búsqueda, por ejemplo con `q=credential`, pero el secreto
+solo queda expuesto porque el fragmento fue indexado antes de que la nota en
+borrador se limpiara.
diff --git a/src/main/resources/explanations/challenge76_hint_fr.adoc b/src/main/resources/explanations/challenge76_hint_fr.adoc
new file mode 100644
index 0000000000..3ffe7b9817
--- /dev/null
+++ b/src/main/resources/explanations/challenge76_hint_fr.adoc
@@ -0,0 +1,16 @@
+Vous pouvez résoudre ce challenge en suivant les étapes suivantes :
+
+1. Ouvrez le point d'accès de recherche non authentifié du magasin vectoriel en mémoire :
+- `http://localhost:8080/rag/search`
+- Il renvoie les quatre fragments de texte indexés au format JSON.
+
+2. Lisez les fragments et cherchez celui qui mentionne un identifiant :
+- Le fragment contenant la « Draft API note » contient l'identifiant du travail
+ d'embedding nocturne.
+
+3. Soumettez cet identifiant comme réponse (tout ce qui suit
+ « uses the credential »).
+
+Vous pouvez aussi affiner la recherche, par exemple avec `q=credential`, mais le
+secret n'est exposé que parce que le fragment a été indexé avant que la note de
+brouillon ne soit nettoyée.
diff --git a/src/main/resources/explanations/challenge76_hint_nl.adoc b/src/main/resources/explanations/challenge76_hint_nl.adoc
new file mode 100644
index 0000000000..82eefc783d
--- /dev/null
+++ b/src/main/resources/explanations/challenge76_hint_nl.adoc
@@ -0,0 +1,15 @@
+Je kunt deze uitdaging oplossen met de volgende stappen:
+
+1. Open het niet-geauthenticeerde zoekendpoint van de vectorstore in het geheugen:
+- `http://localhost:8080/rag/search`
+- Het retourneert alle vier geïndexeerde tekstbrokken als JSON.
+
+2. Lees de brokken en zoek degene die een inloggegeven noemt:
+- De brok met de "Draft API note" bevat het inloggegeven van de nachtelijke
+ embedding-job.
+
+3. Dien dat inloggegeven in als antwoord (alles na "uses the credential ").
+
+Je kunt de zoekopdracht ook verfijnen, bijvoorbeeld met `q=credential`, maar het
+geheim komt alleen bloot te liggen omdat de brok werd geïndexeerd voordat de
+conceptnotitie werd schoongeveegd.
diff --git a/src/main/resources/explanations/challenge76_hint_uk.adoc b/src/main/resources/explanations/challenge76_hint_uk.adoc
new file mode 100644
index 0000000000..e6ad175c99
--- /dev/null
+++ b/src/main/resources/explanations/challenge76_hint_uk.adoc
@@ -0,0 +1,16 @@
+Ви можете розв'язати це завдання, виконавши такі дії:
+
+1. Відкрийте неавтентифікований пошуковий endpoint векторного сховища в пам'яті:
+- `http://localhost:8080/rag/search`
+- Він повертає всі чотири проіндексовані текстові фрагменти у форматі JSON.
+
+2. Прочитайте фрагменти та знайдіть той, що згадує облікові дані:
+- Фрагмент із позначкою "Draft API note" містить облікові дані нічної
+ вбудовувальної задачі.
+
+3. Надішліть ці облікові дані як відповідь (усе, що йде після
+ "uses the credential ").
+
+Ви також можете уточнити запит, наприклад `q=credential`, але секрет
+виявляється лише тому, що фрагмент було проіндексовано до того, як чернетку
+очистили.
diff --git a/src/main/resources/explanations/challenge76_nl.adoc b/src/main/resources/explanations/challenge76_nl.adoc
new file mode 100644
index 0000000000..7aed59a75b
--- /dev/null
+++ b/src/main/resources/explanations/challenge76_nl.adoc
@@ -0,0 +1,30 @@
+=== Uitdaging 76: Vind het geheim in de RAG-vectorstore
+
+Pipelines voor retrieval augmented generation (RAG) beantwoorden vragen door eerst
+relevante passages op te zoeken in een vectorstore en die passages vervolgens aan
+een groot taalmodel te voeren. Om die store op te bouwen, worden documenten in
+brokken gesplitst, omgezet in embeddings en geïndexeerd. De brokken zelf blijven
+leesbaar: een zoekendpoint retourneert de oorspronkelijke tekst van elke brok die
+relevant wordt geacht.
+
+Dat is precies wat er in deze uitdaging misging. Een ontwikkelteam.indexeerde zijn runbooks, vergadernotulen en concept-API-documentatie in een
+vectorstore die vanuit de toepassing bereikbaar is zonder enige authenticatie:
+
+`GET /rag/search`
+
+Het endpoint accepteert een optionele queryparameter `q` en retourneert de
+geïndexeerde tekstbrokken die de store als relevant voor die query beoordeelt.
+Zonder query retourneert het elke geïndexeerde brok. Eén van die brokken werd
+geïndexeerd terwijl hij nog een conceptnotitie bevatte met het inloggegeven van
+de nachtelijke embedding-job.
+
+Doorblader de geïndexeerde brokken via het endpoint en dien het inloggegeven dat
+je daarin vindt in.
+
+[NOTE]
+====
+Het endpoint en alle inloggegevens in deze uitdaging zijn fictief. Er wordt niets
+over het netwerk verstuurd: de toepassing zet de brokken om in embeddings en
+indexeert ze bij het starten in zijn eigen vectorindex in het geheugen; alleen
+die index wordt doorzocht.
+====
diff --git a/src/main/resources/explanations/challenge76_reason.adoc b/src/main/resources/explanations/challenge76_reason.adoc
new file mode 100644
index 0000000000..540775b1bb
--- /dev/null
+++ b/src/main/resources/explanations/challenge76_reason.adoc
@@ -0,0 +1,44 @@
+*Why a RAG vector store leaks the secrets inside the documents it indexed*
+
+Retrieval augmented generation stores the text of documents so that a search can
+return it later. That text is not transformed away by the embedding step: the
+embeddings are computed from the chunks, but the chunks themselves remain fully
+readable and are returned verbatim by every search. Index a document that contains
+a secret and the secret becomes part of the retrieval corpus.
+
+The failure in this challenge is typical:
+
+- Documents are indexed in bulk, without a review of what they contain. Draft
+ notes, runbooks and meeting minutes are exactly the kind of informal text that
+ collects credentials "just for now".
+- The development vector store was opened up "so that everybody can try RAG",
+ without authentication on the search endpoint. Retrieval interfaces are rarely
+ treated with the same care as database logins, even though they return the same
+ content.
+- Chunking splits documents at arbitrary boundaries, so a secret often ends up in
+ the middle of an innocent-looking chunk. Scanning only the first lines of a
+ document is not enough.
+- Vector stores keep their metadata: which document a chunk came from, when it was
+ indexed and who uploaded it. That metadata tells an attacker exactly which leak
+ is worth exfiltrating.
+
+----
+What to do instead:
+
+- Treat the vector store as production data: require authentication on every
+ retrieval endpoint and authorize per collection or tenant.
+- Scan and scrub documents before they are indexed, the same way repositories and
+ tickets are scanned for secrets.
+- Keep credentials out of the corpus entirely: reference them by name and resolve
+ them at runtime, so that re-indexing cannot leak them.
+- Log and monitor retrieval queries; an unusual query pattern is often the first
+ sign that someone is browsing the corpus for secrets.
+- Rotate any credential that was ever part of an indexed document, even after the
+ document is deleted: copies of the chunk may live in backups and replicas.
+----
+
+[NOTE]
+====
+Anything that would be unacceptable in a git repository is unacceptable in a
+vector store. An indexed chunk is a copy of the document, with all of its secrets.
+====
diff --git a/src/main/resources/explanations/challenge76_reason_de.adoc b/src/main/resources/explanations/challenge76_reason_de.adoc
new file mode 100644
index 0000000000..e8d6a25988
--- /dev/null
+++ b/src/main/resources/explanations/challenge76_reason_de.adoc
@@ -0,0 +1,46 @@
+*Warum ein RAG-Vektorstore die Geheimnisse der von ihm indizierten Dokumente leakt*
+
+Retrieval-Augmented-Generation speichert den Text von Dokumenten, damit eine Suche ihn
+später zurückgeben kann. Dieser Text verschwindet nicht durch den Embedding-Schritt:
+Die Embeddings werden aus den Chunks berechnet, aber die Chunks selbst bleiben vollständig
+lesbar und werden von jeder Suche wörtlich zurückgegeben. Indiziere ein Dokument, das ein
+Geheimnis enthält, und das Geheimnis wird Teil des Retrieval-Korpus.
+
+Der Fehler in dieser Challenge ist typisch:
+
+- Dokumente werden in großen Mengen indiziert, ohne zu prüfen, was sie enthalten.
+ Entwurfsnotizen, Runbooks und Protokolle sind genau die Art informeller Text, der
+ Zugangscredentials "nur für einen Moment" sammelt.
+- Der Entwicklungs-Vektorstore wurde "damit jeder RAG ausprobieren kann" geöffnet,
+ ohne Authentifizierung auf dem Suchendpoint. Retrieval-Schnittstellen werden selten
+ mit derselben Sorgfalt behandelt wie Datenbank-Logins, obwohl sie denselben Inhalt
+ zurückgeben.
+- Chunking teilt Dokumente an beliebigen Grenzen, sodass ein Geheimnis oft in der
+ Mitte eines harmlos aussehenden Chunks landet. Nur die ersten Zeilen eines Dokuments
+ zu scannen, reicht nicht aus.
+- Vektorstores bewahren ihre Metadaten: aus welchem Dokument ein Chunk stammt, wann
+ er indiziert wurde und wer ihn hochgeladen hat. Diese Metadaten verraten einem
+ Angreifer genau, welcher Leak sich lohnt, exfiltriert zu werden.
+
+----
+Was du stattdessen tun solltest:
+
+- Behandle den Vektorstore wie Produktionsdaten: Verlange Authentifizierung auf jedem
+ Retrieval-Endpoint und autorisiere pro Collection oder Tenant.
+- Scanne und bereinige Dokumente, bevor sie indiziert werden, auf dieselbe Weise wie
+ Repositories und Tickets nach Geheimnissen gescannt werden.
+- Halte Zugangscredentials vollständig aus dem Korpus: Referenziere sie über Namen
+ und löse sie zur Laufzeit auf, damit eine Neuindizierung sie nicht leaken kann.
+- Protokolliere und überwache Retrieval-Queries; ein ungewöhnliches Query-Muster ist
+ oft das erste Zeichen, dass jemand den Korpus nach Geheimnissen durchsucht.
+- Rotiere jedes Zugangscredential, das jemals Teil eines indizierten Dokuments war,
+ auch nachdem das Dokument gelöscht wurde: Kopien des Chunks können in Backups und
+ Replikaten weiterleben.
+----
+
+[NOTE]
+====
+Alles, was in einem Git-Repository inakzeptabel wäre, ist in einem Vektorstore
+ebenfalls inakzeptabel. Ein indizierter Chunk ist eine Kopie des Dokuments, mit all
+seinen Geheimnissen.
+====
diff --git a/src/main/resources/explanations/challenge76_reason_es.adoc b/src/main/resources/explanations/challenge76_reason_es.adoc
new file mode 100644
index 0000000000..01e05fbf34
--- /dev/null
+++ b/src/main/resources/explanations/challenge76_reason_es.adoc
@@ -0,0 +1,48 @@
+*Por qué un almacén vectorial RAG filtra los secretos de los documentos que indexó*
+
+La generación aumentada por recuperación almacena el texto de los documentos para
+que una búsqueda pueda devolverlo después. Ese texto no desaparece por el paso de
+embedding: los embeddings se calculan a partir de los fragmentos, pero los
+fragmentos en sí permanecen totalmente legibles y cada búsqueda los devuelve
+literalmente. Indexa un documento que contenga un secreto y el secreto pasa a
+formar parte del corpus de recuperación.
+
+El fallo de este desafío es típico:
+
+- Los documentos se indexan en masa, sin revisar qué contienen. Las notas en
+ borrador, los runbooks y las actas de reunión son exactamente el tipo de texto
+ informal que acumula credenciales "solo por ahora".
+- El almacén vectorial de desarrollo se abrió "para que todos pudieran probar RAG",
+ sin autenticación en el endpoint de búsqueda. A las interfaces de recuperación
+ rara vez se les presta el mismo cuidado que a los inicios de sesión de bases de
+ datos, aunque devuelven el mismo contenido.
+- El chunking divide los límites de los documentos de forma arbitraria, por lo que
+ un secreto a menudo termina en medio de un fragmento que parece inofensivo.
+ Escanear solo las primeras líneas de un documento no es suficiente.
+- Los almacenes vectoriales conservan sus metadatos: de qué documento procede un
+ fragmento, cuándo se indexó y quién lo subió. Esos metadatos le dicen a un
+ atacante exactamente qué filtración vale la pena exfiltrar.
+
+----
+Qué hacer en su lugar:
+
+- Trata el almacén vectorial como datos de producción: exige autenticación en cada
+ endpoint de recuperación y autoriza por colección o inquilino.
+- Escanea y limpia los documentos antes de indexarlos, de la misma forma en que se
+ escanean repositorios y tickets en busca de secretos.
+- Mantén las credenciales completamente fuera del corpus: refléjalas por nombre y
+ resuélvelas en tiempo de ejecución, para que la reindexación no pueda filtrarlas.
+- Registra y supervisa las consultas de recuperación; un patrón de consultas inusual
+ suele ser la primera señal de que alguien está explorando el corpus en busca de
+ secretos.
+- Rota cualquier credencial que haya formado parte de un documento indexado, incluso
+ después de borrar el documento: pueden quedar copias del fragmento en copias de
+ seguridad y réplicas.
+----
+
+[NOTE]
+====
+Cualquier cosa que sería inaceptable en un repositorio git también es inaceptable
+en un almacén vectorial. Un fragmento indexado es una copia del documento, con
+todos sus secretos.
+====
diff --git a/src/main/resources/explanations/challenge76_reason_fr.adoc b/src/main/resources/explanations/challenge76_reason_fr.adoc
new file mode 100644
index 0000000000..af8ac26c4e
--- /dev/null
+++ b/src/main/resources/explanations/challenge76_reason_fr.adoc
@@ -0,0 +1,48 @@
+*Pourquoi un magasin vectoriel RAG fuit les secrets des documents qu'il a indexés*
+
+La génération augmentée par récupération stocke le texte des documents afin qu'une
+recherche puisse le renvoyer plus tard. Ce texte ne disparaît pas grâce à l'étape
+d'embedding : les embeddings sont calculés à partir des fragments, mais les
+fragments eux-mêmes restent entièrement lisibles et chaque recherche les renvoie
+verbatim. Indexez un document qui contient un secret et le secret devient partie
+intégrante du corpus de récupération.
+
+La faille de ce challenge est typique :
+
+- Les documents sont indexés en masse, sans examen de leur contenu. Les notes de
+ brouillon, les runbooks et les comptes rendus de réunion sont exactement le type
+ de texte informel qui accumule des identifiants « juste pour le moment ».
+- Le magasin vectoriel de développement a été ouvert « pour que tout le monde
+ puisse essayer le RAG », sans authentification sur le point d'accès de recherche.
+ Les interfaces de récupération sont rarement traitées avec le même soin que les
+ connexions aux bases de données, alors qu'elles renvoient le même contenu.
+- Le découpage en fragments divise les documents à des frontières arbitraires, si
+ bien qu'un secret se retrouve souvent au milieu d'un fragment anodin. Scanner
+ uniquement les premières lignes d'un document ne suffit pas.
+- Les magasins vectoriels conservent leurs métadonnées : le document d'origine
+ d'un fragment, sa date d'indexation et son téléverseur. Ces métadonnées indiquent
+ exactement à un attaquant quelle fuite vaut la peine d'être exfiltrée.
+
+----
+Ce qu'il faut faire à la place :
+
+- Traitez le magasin vectoriel comme des données de production : exigez une
+ authentification sur chaque point d'accès de récupération et autorisez par
+ collection ou locataire.
+- Analysez et nettoyez les documents avant de les indexer, de la même manière que
+ les dépôts et les tickets sont analysés à la recherche de secrets.
+- Gardez les identifiants entièrement hors du corpus : référencez-les par leur nom
+ et résolvez-les à l'exécution, afin qu'une réindexation ne puisse pas les révéler.
+- Journalisez et surveillez les requêtes de récupération ; un motif de requêtes
+ inhabituel est souvent le premier signe que quelqu'un parcourt le corpus à la
+ recherche de secrets.
+- Révoquez et remplacez tout identifiant ayant fait partie d'un document indexé,
+ même une fois le document supprimé : des copies du fragment peuvent subsister
+ dans les sauvegardes et les répliques.
+----
+
+[NOTE]
+====
+Tout ce qui serait inacceptable dans un dépôt git l'est aussi dans un magasin
+vectoriel. Un fragment indexé est une copie du document, avec tous ses secrets.
+====
diff --git a/src/main/resources/explanations/challenge76_reason_nl.adoc b/src/main/resources/explanations/challenge76_reason_nl.adoc
new file mode 100644
index 0000000000..e2417888e8
--- /dev/null
+++ b/src/main/resources/explanations/challenge76_reason_nl.adoc
@@ -0,0 +1,45 @@
+*Waarom een RAG-vectorstore de geheimen lekt van de documenten die hij indexeerde*
+
+Retrieval augmented generation slaat de tekst van documenten op zodat een zoekopdracht
+die tekst later kan retourneren. Die tekst verdwijnt niet door de embedding-stap: de
+embeddings worden uit de brokken berekend, maar de brokken zelf blijven volledig
+leesbaar en worden door elke zoekopdracht letterlijk teruggegeven. Indexeer een
+document dat een geheim bevat en het geheim wordt onderdeel van de retrieval-corpus.
+
+De fout in deze uitdaging is typisch:
+
+- Documenten worden in bulk geïndexeerd, zonder controle van wat ze bevatten.
+ Conceptnotities, runbooks en vergadernotulen zijn precies het soort informele
+ tekst dat inloggegevens verzamelt "voor even".
+- De ontwikkelvectorstore werd opengesteld "zodat iedereen RAG kan proberen", zonder
+ authenticatie op het zoekendpoint. Retrieval-interfaces worden zelden met dezelfde
+ zorg behandeld als database-inloggegevens, terwijl ze dezelfde inhoud teruggeven.
+- Chunking splitst documenten op willekeurige grenzen, waardoor een geheim vaak in
+ het midden van een onschuldig uitziende brok belandt. Alleen de eerste regels van
+ een document scannen is niet genoeg.
+- Vectorstores bewaren hun metadata: uit welk document een brok komt, wanneer die werd
+ geïndexeerd en wie die heeft geüpload. Die metadata vertelt een aanvaller precies
+ welke lek het waard is om te exfiltreren.
+
+----
+Wat je beter kunt doen:
+
+- Behandel de vectorstore als productiedata: eis authenticatie op elk
+ retrieval-endpoint en autoriseer per collectie of tenant.
+- Scan en reinig documenten voordat ze worden geïndexeerd, op dezelfde manier waarop
+ repositories en tickets op geheimen worden gescand.
+- Houd inloggegevens volledig buiten het corpus: verwijs ernaar op naam en los ze op
+ runtime op, zodat herindexeren ze niet kan lekken.
+- Log en monitor retrieval-query's; een ongebruikelijk querypatroon is vaak het eerste
+ teken dat iemand door het corpus bladert op zoek naar geheimen.
+- Roteer elk inloggegeven dat ooit deel uitmaakte van een geïndexeerd document, ook
+ nadat het document is verwijderd: kopieën van de brok kunnen in backups en replicas
+ blijven bestaan.
+----
+
+[NOTE]
+====
+Alles wat in een git-repository onacceptabel zou zijn, is in een vectorstore ook
+onacceptabel. Een geïndexeerde brok is een kopie van het document, met al zijn
+geheimen.
+====
diff --git a/src/main/resources/explanations/challenge76_reason_uk.adoc b/src/main/resources/explanations/challenge76_reason_uk.adoc
new file mode 100644
index 0000000000..c20df99ab2
--- /dev/null
+++ b/src/main/resources/explanations/challenge76_reason_uk.adoc
@@ -0,0 +1,45 @@
+*Чому векторне сховище RAG витікає секрети документів, які воно проіндексувало*
+
+Retrieval augmented generation зберігає текст документів, щоб пошук міг повернути
+його пізніше. Цей текст не зникає на етапі embedding: embedding'и обчислюються з
+фрагментів, але самі фрагменти залишаються повністю читабельними, і кожен пошук
+повертає їх дослівно. Проіндексуйте документ, що містить секрет, і секрет стане
+частиною retrieval-корпусу.
+
+Помилка в цьому завданні є типовою:
+
+- Документи індексуються масово, без перевірки їхнього вмісту. Чернетки, runbook'и
+ та нотатки з нарад — саме той неформальний текст, у якому накопичуються облікові
+ дані «тимчасово».
+- Векторне сховище розробки відкрили «щоб усі могли спробувати RAG», без
+ автентифікації на пошуковому endpoint. До retrieval-інтерфейсів рідко ставляться
+ з такою ж увагою, як до входу в базу даних, хоча вони повертають той самий
+ вміст.
+- Розбиття на фрагменти ділить документи за довільними межами, тому секрет часто
+ опиняється в середині безневинного на вигляд фрагмента. Сканування лише перших
+ рядків документа недостатньо.
+- Векторні сховища зберігають свої метадані: з якого документа походить фрагмент,
+ коли його проіндексували і хто його завантажив. Ці метадані підказують
+ нападнику саме той витік, який варто екфільтрувати.
+
+----
+Що краще робити натомість:
+
+- Ставтеся до векторного сховища як до виробничих даних: вимагайте автентифікації
+ на кожному retrieval-endpoint і авторизуйте за колекцією або орендарем.
+- Скануйте та очищуйте документи перед індексацією — так само, як репозиторії та
+ тікети сканують на секрети.
+- Тримайте облікові дані повністю за межами корпусу: посилайтеся на них за іменем
+ і розв'язуйте їх під час виконання, щоб повторна індексація не могла їх витекти.
+- Реєструйте та моніторте retrieval-запити; незвичний шаблон запитів часто є
+ першою ознакою того, що хтось шукає секрети в корпусі.
+- Ротуйте будь-які облікові дані, які коли-небудь були частиною проіндексованого
+ документа, навіть після його видалення: копії фрагмента можуть лишитися в
+ резервних копіях та репліках.
+----
+
+[NOTE]
+====
+Усе, що було б неприйнятно в git-репозиторії, неприйнятно і у векторному сховищі.
+Проіндексований фрагмент — це копія документа з усіма його секретами.
+====
diff --git a/src/main/resources/explanations/challenge76_uk.adoc b/src/main/resources/explanations/challenge76_uk.adoc
new file mode 100644
index 0000000000..88dbed7e4f
--- /dev/null
+++ b/src/main/resources/explanations/challenge76_uk.adoc
@@ -0,0 +1,31 @@
+=== Завдання 76: Знайдіть секрет у векторному сховищі RAG
+
+Пайплайни retrieval augmented generation (RAG) відповідають на запити, спочатку
+шуючи релевантні уривки у векторному сховищі, а потім передаючи ці уривки
+великій мовній моделі. Щоб побудувати таке сховище, документи розбиваються на
+фрагменти, перетворюються на embedding'и та індексуються. Самі фрагменти
+залишаються читабельними: пошуковий endpoint повертає оригінальний текст кожного
+фрагмента, який вважається релевантним.
+
+Саме це й пішло не так у цьому завданні. Команда розробки проіндексувала свої
+runbook'и, нотатки з нарад та чернетки API-документації у векторне сховище, яке
+доступне з програми без жодної автентифікації:
+
+`GET /rag/search`
+
+Endpoint приймає необов'язковий параметр запиту `q` і повертає проіндексовані
+текстові фрагменти, які сховище вважає релевантними для нього. Без запиту він
+повертає всі проіндексовані фрагменти. Один із фрагментів було проіндексовано,
+коли він ще містив чернетку з обліковими даними нічної вбудовувальної (embedding)
+задачі.
+
+Перегляньте проіндексовані фрагменти через endpoint і надішліть знайдені там
+облікові дані.
+
+[NOTE]
+====
+Endpoint і всі облікові дані в цьому завданні є вигаданими. Нічого не
+передається через мережу: програма перетворює фрагменти на embedding'и та
+індексує їх у власному векторному індексі в пам'яті під час запуску, а потім
+шукає лише в ньому.
+====
diff --git a/src/main/resources/wrong-secrets-configuration.yaml b/src/main/resources/wrong-secrets-configuration.yaml
index b5861dca00..bc918c4a10 100644
--- a/src/main/resources/wrong-secrets-configuration.yaml
+++ b/src/main/resources/wrong-secrets-configuration.yaml
@@ -1145,3 +1145,16 @@ configurations:
category: *ci_cd
ctf:
enabled: true
+
+ - name: Challenge 76
+ short-name: "challenge-76"
+ sources:
+ - class-name: "org.owasp.wrongsecrets.challenges.docker.Challenge76"
+ explanation: "explanations/challenge76.adoc"
+ hint: "explanations/challenge76_hint.adoc"
+ reason: "explanations/challenge76_reason.adoc"
+ environments: *all_envs
+ difficulty: *easy
+ category: *ai
+ ctf:
+ enabled: true
diff --git a/src/test/java/org/owasp/wrongsecrets/ChallengeContentLocalizationCoverageTest.java b/src/test/java/org/owasp/wrongsecrets/ChallengeContentLocalizationCoverageTest.java
new file mode 100644
index 0000000000..1fa64bf4f7
--- /dev/null
+++ b/src/test/java/org/owasp/wrongsecrets/ChallengeContentLocalizationCoverageTest.java
@@ -0,0 +1,76 @@
+package org.owasp.wrongsecrets;
+
+import static org.assertj.core.api.Assertions.assertThat;
+
+import java.util.HashSet;
+import java.util.List;
+import java.util.Objects;
+import java.util.Set;
+import java.util.stream.Stream;
+import org.junit.jupiter.api.Test;
+import org.owasp.wrongsecrets.definitions.ChallengeDefinitionsConfiguration;
+import org.owasp.wrongsecrets.definitions.Sources.ChallengeSource;
+import org.owasp.wrongsecrets.definitions.Sources.TextWithFileLocation;
+import org.springframework.beans.factory.annotation.Autowired;
+import org.springframework.boot.test.context.SpringBootTest;
+import org.springframework.core.io.ClassPathResource;
+
+/**
+ * Verifies that every challenge content file referenced by the configuration (explanation, hint and
+ * reason of all 77 challenges, including cloud and limited-hint variants) exists in English, and
+ * that the content of challenge 76 is translated for each supported locale (nl, de, es, fr, uk).
+ */
+@SpringBootTest
+class ChallengeContentLocalizationCoverageTest {
+
+ private static final List SUPPORTED_LANGUAGES = List.of("nl", "de", "es", "fr", "uk");
+
+ private static final List CHALLENGE_76_CONTENT_FILES =
+ List.of(
+ "explanations/challenge76.adoc",
+ "explanations/challenge76_hint.adoc",
+ "explanations/challenge76_reason.adoc");
+
+ @Autowired private ChallengeDefinitionsConfiguration definitions;
+
+ @Test
+ void everyReferencedChallengeContentFileExistsInEnglish() {
+ Set referencedFiles = new HashSet<>();
+ for (var definition : definitions.challenges()) {
+ for (ChallengeSource source : definition.sources()) {
+ Stream.of(source.explanation(), source.hint(), source.reason(), source.hintLimited())
+ .filter(Objects::nonNull)
+ .map(TextWithFileLocation::fileName)
+ .filter(Objects::nonNull)
+ .filter(fileName -> !fileName.isBlank())
+ .forEach(referencedFiles::add);
+ }
+ }
+
+ assertThat(definitions.challenges()).hasSize(77);
+ assertThat(referencedFiles).hasSize(254);
+
+ for (String fileName : referencedFiles) {
+ assertThat(new ClassPathResource(fileName).exists())
+ .as("English content file %s should exist", fileName)
+ .isTrue();
+ }
+ }
+
+ @Test
+ void challenge76ContentShouldBeTranslatedForAllSupportedLocales() {
+ for (String fileName : CHALLENGE_76_CONTENT_FILES) {
+ assertThat(new ClassPathResource(fileName).exists())
+ .as("English content file %s should exist", fileName)
+ .isTrue();
+
+ String base = fileName.substring(0, fileName.lastIndexOf('.'));
+ for (String language : SUPPORTED_LANGUAGES) {
+ String localizedFileName = base + "_" + language + ".adoc";
+ assertThat(new ClassPathResource(localizedFileName).exists())
+ .as("Localized file %s for language %s should exist", fileName, language)
+ .isTrue();
+ }
+ }
+ }
+}
diff --git a/src/test/java/org/owasp/wrongsecrets/challenges/docker/Challenge76Test.java b/src/test/java/org/owasp/wrongsecrets/challenges/docker/Challenge76Test.java
new file mode 100644
index 0000000000..c28468692c
--- /dev/null
+++ b/src/test/java/org/owasp/wrongsecrets/challenges/docker/Challenge76Test.java
@@ -0,0 +1,125 @@
+package org.owasp.wrongsecrets.challenges.docker;
+
+import static org.assertj.core.api.Assertions.assertThat;
+import static org.hamcrest.CoreMatchers.containsString;
+import static org.hamcrest.CoreMatchers.not;
+import static org.springframework.test.web.servlet.request.MockMvcRequestBuilders.get;
+import static org.springframework.test.web.servlet.result.MockMvcResultMatchers.content;
+import static org.springframework.test.web.servlet.result.MockMvcResultMatchers.status;
+
+import org.junit.jupiter.api.Test;
+import org.owasp.wrongsecrets.Challenges;
+import org.owasp.wrongsecrets.challenges.Spoiler;
+import org.springframework.beans.factory.annotation.Autowired;
+import org.springframework.boot.test.context.SpringBootTest;
+import org.springframework.boot.webmvc.test.autoconfigure.AutoConfigureMockMvc;
+import org.springframework.test.web.servlet.MockMvc;
+
+@SpringBootTest
+@AutoConfigureMockMvc
+class Challenge76Test {
+
+ @Autowired private Challenges challenges;
+
+ @Autowired private MockMvc mvc;
+
+ @Test
+ void spoilerShouldReturnTheIndexedDevelopmentSecret() {
+ var challenge = new Challenge76();
+
+ assertThat(challenge.spoiler()).isEqualTo(new Spoiler(Challenge76.DEVELOPMENT_SECRET));
+ }
+
+ @Test
+ void answerCorrectShouldReturnTrueForCorrectAnswer() {
+ var challenge = new Challenge76();
+
+ assertThat(challenge.answerCorrect(Challenge76.DEVELOPMENT_SECRET)).isTrue();
+ assertThat(challenge.answerCorrect(challenge.spoiler().solution())).isTrue();
+ }
+
+ @Test
+ void answerCorrectShouldReturnFalseForIncorrectAnswer() {
+ var challenge = new Challenge76();
+
+ assertThat(challenge.answerCorrect("wronganswer")).isFalse();
+ assertThat(challenge.answerCorrect("")).isFalse();
+ assertThat(challenge.answerCorrect(null)).isFalse();
+ assertThat(challenge.answerCorrect("vector-store-dev-secret")).isFalse();
+ }
+
+ @Test
+ void searchShouldReturnTheChunkThatContainsTheSecret() {
+ var challenge = new Challenge76();
+
+ var results = challenge.search("credential");
+
+ assertThat(results).hasSize(1);
+ assertThat(results.getFirst().text()).contains(Challenge76.DEVELOPMENT_SECRET);
+ }
+
+ @Test
+ void searchWithoutQueryShouldReturnAllIndexedChunks() {
+ var challenge = new Challenge76();
+
+ assertThat(challenge.search(null)).hasSize(4);
+ assertThat(challenge.search(" ")).hasSize(4);
+ }
+
+ @Test
+ void searchShouldMatchCaseInsensitivelyAndReturnEmptyForUnknownQueries() {
+ var challenge = new Challenge76();
+
+ assertThat(challenge.search("RECALL")).hasSize(1);
+ assertThat(challenge.search("RECALL").getFirst().text()).contains("0.78");
+ assertThat(challenge.search("does-not-exist")).isEmpty();
+ }
+
+ @Test
+ void searchShouldReturnTheSecretChunkForQueriesAboutIt() {
+ var challenge = new Challenge76();
+
+ var results = challenge.search("nightly embedding credential");
+
+ assertThat(results).hasSize(1);
+ assertThat(results.getFirst().text()).contains(Challenge76.DEVELOPMENT_SECRET);
+ }
+
+ @Test
+ void searchShouldMatchIndexedTermsInsteadOfArbitrarySubstrings() {
+ var challenge = new Challenge76();
+
+ assertThat(challenge.search("ecret")).isEmpty();
+ assertThat(challenge.search("e5c91a7b")).hasSize(1);
+ assertThat(challenge.search("e5c91a7b").getFirst().text())
+ .contains(Challenge76.DEVELOPMENT_SECRET);
+ }
+
+ @Test
+ void searchEndpointShouldBeUnauthenticatedAndLeakTheSecret() throws Exception {
+ mvc.perform(get("/rag/search"))
+ .andExpect(status().isOk())
+ .andExpect(content().string(containsString(Challenge76.DEVELOPMENT_SECRET)));
+ }
+
+ @Test
+ void searchEndpointShouldFilterOnTheQueryParameter() throws Exception {
+ mvc.perform(get("/rag/search").param("q", "recall"))
+ .andExpect(status().isOk())
+ .andExpect(content().string(containsString("0.78")))
+ .andExpect(content().string(not(containsString(Challenge76.DEVELOPMENT_SECRET))));
+ }
+
+ @Test
+ void challenge76ShouldBeRegisteredAndSolveable() {
+ var definition = challenges.findByShortName("challenge-76");
+
+ assertThat(definition).isPresent();
+ assertThat(challenges.getChallenge(definition.get())).hasSize(1);
+ var challenge = challenges.getChallenge(definition.get()).getFirst();
+
+ assertThat(challenge).isInstanceOf(Challenge76.class);
+ assertThat(challenge.spoiler().solution()).isNotEmpty();
+ assertThat(challenge.answerCorrect(challenge.spoiler().solution())).isTrue();
+ }
+}