feat(ai-sdk): legal-corpus structure endpoint + coverage page

Expose GET /sdk/v1/rag/legal-corpus, which scrolls the eur-lex legal corpus (filtered to a few hundred points regardless of total size) and aggregates each ingested act's composition: distinct articles, annexes, recitals and chunk count. Surface it as a new section on /sdk/coverage so the ingested corpus is no longer a black box — a developer SEES what each act actually contains, not only its name. Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
2026-06-23 19:47:17 +02:00
parent b83c3e6e00
commit 4c99773fa1
6 changed files with 352 additions and 3 deletions
@@ -206,3 +206,32 @@ func (h *RAGHandlers) HandleScrollChunks(c *gin.Context) {
 		"total":       len(chunks),
 	})
 }
+
+// LegalCorpusStructure returns the composition (distinct articles, annexes,
+// recitals + chunk count) of every ingested eur-lex legal act, so the coverage
+// page can show WHAT was ingested instead of just the act name.
+// GET /sdk/v1/rag/legal-corpus
+func (h *RAGHandlers) LegalCorpusStructure(c *gin.Context) {
+	acts, err := h.ragClient.CorpusStructure(c.Request.Context())
+	if err != nil {
+		c.JSON(http.StatusInternalServerError, gin.H{"error": "failed to aggregate legal corpus: " + err.Error()})
+		return
+	}
+
+	arts, anns, recs := 0, 0, 0
+	for _, a := range acts {
+		arts += a.Articles
+		anns += a.Annexes
+		recs += a.Recitals
+	}
+
+	c.JSON(http.StatusOK, gin.H{
+		"regulations": acts,
+		"totals": gin.H{
+			"regulations": len(acts),
+			"articles":    arts,
+			"annexes":     anns,
+			"recitals":    recs,
+		},
+	})
+}