compaction de contexte, filtre/pagination/suppression des conversations

- compaction automatique du contexte LLM quand la taille atteint le seuil
  (contexte max - taille de réponse max), par résumé de l'historique ancien
- filtre textuel (?q=) et pagination (20/page) de l'historique des conversations
- correctif ORM : offset de pagination en base 0
- suppression d'une conversation : messages associés supprimés, session en
  cours annulée, popup de confirmation (ConfirmDialog)
- titres de conversation nettoyés du markdown
- styles des listes markdown, variante danger du bouton
- go.mod/go.sum : dépendances go-rod et stealth manquantes du commit précédent
This commit is contained in:
fabien committed 2026-10-05 23:46:12 +02:00
1 parent a970bfa991
commit 51bbb45b32
86 files changed
+770 -113

No files matched your search

+105
View File
@@ -0,0 +1,105 @@
package chat
import (
"context"
"fmt"
"log/slog"
"strings"
"trankilou.fr/lassistanoque/backend/internal/domain"
)
const (
// charsPerToken : estimation heuristique, ~4 caractères par token pour du
// texte européen. Suffisant pour décider d'une compaction.
charsPerToken = 4
// defaultContextMaxTokens : taille de contexte du modèle quand la
// configuration ne la précise pas.
defaultContextMaxTokens = 32000
// defaultMaxResponseTokens : réserve à ménager pour la réponse du modèle.
defaultMaxResponseTokens = 4096
)
// estimateTokens estime le nombre de tokens d'un texte.
func estimateTokens(text string) int {
if text == "" {
return 0
}
tokens := len(text) / charsPerToken
if len(text)%charsPerToken > 0 {
tokens++
}
return tokens
}
// contextTokens estime la taille en tokens du prompt complet envoyé au LLM.
func (s *ChatSession) contextTokens(systemPrompt string) int {
total := estimateTokens(systemPrompt)
for _, m := range s.messages {
total += estimateTokens(m.Content) + estimateTokens(m.ToolCallsJson)
}
return total
}
// shouldCompact indique si la taille du contexte a atteint la limite au-delà
// de laquelle il ne reste plus assez de place pour une réponse complète :
// contexte max diminué de la taille de réponse max.
func (s *ChatSession) shouldCompact(systemPrompt string) bool {
return s.contextTokens(systemPrompt) >= s.contextMaxTokens-s.maxResponseTokens
}
// compactContext remplace l'historique ancien par un résumé généré par le LLM.
// Les échanges récents (à partir du dernier message utilisateur) sont conservés
// intacts : un point de coupure sur un message utilisateur ne peut jamais
// séparer un message assistant porteur d'appels d'outils de ses réponses tool,
// ce qui garderait un historique incohérent pour l'API.
// La compaction ne porte que sur la session en mémoire : l'historique complet
// reste intact en base.
func (s *ChatSession) compactContext(ctx context.Context) {
cut := -1
for i := len(s.messages) - 1; i >= 0; i-- {
if s.messages[i].Role == string(domain.RoleUser) {
cut = i
break
}
}
// rien à compacter : l'historique commence déjà au dernier message utilisateur
if cut <= 0 {
return
}
head := s.messages[:cut]
tail := s.messages[cut:]
var sb strings.Builder
for _, m := range head {
fmt.Fprintf(&sb, "%s: %s\n", m.Role, m.Content)
}
summary, err := s.llmEngine.Generate(ctx, s.provider, s.modelID, []*domain.Message{
{
Role: string(domain.RoleUser),
Content: "Résume de façon concise et factuelle la conversation suivante. " +
"Conserve les informations essentielles : demandes, décisions, résultats, fichiers ou liens mentionnés. " +
"Réponds en texte brut, sans markdown.\n\n" + sb.String(),
},
})
if err != nil {
slog.Error("compactContext.Generate", "error", err)
return
}
summaryMessage := &domain.Message{
Role: string(domain.RoleUser),
Content: "Résumé de la conversation précédente (les messages plus anciens ont été compactés) :\n" +
summary,
}
s.messages = append([]*domain.Message{summaryMessage}, tail...)
slog.Info("context compacted",
"removedMessages", len(head),
"keptMessages", len(tail)+1,
)
}
@@ -0,0 +1,97 @@
package chat
import (
"context"
"strings"
"testing"
"trankilou.fr/lassistanoque/backend/internal/domain"
)
type fakeEngine struct{ summary string }
func (f *fakeEngine) ListProviderTypes() []domain.Item { return nil }
func (f *fakeEngine) ListModelsFromProvider(ctx context.Context, p *domain.Provider) ([]string, error) {
return nil, nil
}
func (f *fakeEngine) Stream(ctx context.Context, p *domain.Provider, m string, params *domain.LLMParams, msgs []*domain.Message) *domain.Message {
return &domain.Message{}
}
func (f *fakeEngine) Generate(ctx context.Context, p *domain.Provider, m string, msgs []*domain.Message) (string, error) {
return f.summary, nil
}
func mk(role, content string) *domain.Message {
return &domain.Message{Role: role, Content: content}
}
func TestEstimateTokens(t *testing.T) {
if got := estimateTokens(""); got != 0 {
t.Errorf("empty = %d", got)
}
if got := estimateTokens("abcd"); got != 1 {
t.Errorf("abcd = %d", got)
}
if got := estimateTokens("abcde"); got != 2 {
t.Errorf("abcde = %d", got)
}
}
func TestShouldCompactThreshold(t *testing.T) {
s := &ChatSession{
messages: []*domain.Message{mk(string(domain.RoleUser), strings.Repeat("a", 320))}, // 80 tokens
contextMaxTokens: 100,
maxResponseTokens: 20,
}
if !s.shouldCompact("system") { // 80 + 1 >= 100-20
t.Errorf("should compact at contextMax - maxResponse boundary")
}
s.maxResponseTokens = 40
s.contextMaxTokens = 130 // seuil 130-40=90 > 82 : pas de compaction
if s.shouldCompact("system") {
t.Errorf("should not compact below threshold")
}
}
func TestCompactContext(t *testing.T) {
s := &ChatSession{
provider: &domain.Provider{},
llmEngine: &fakeEngine{summary: "resume des echanges"},
contextMaxTokens: 100,
maxResponseTokens: 10,
messages: []*domain.Message{
mk(string(domain.RoleUser), "ancienne question 1"),
mk(string(domain.RoleAssistant), "ancienne reponse 1"),
mk(string(domain.RoleUser), "question courante"),
mk(string(domain.RoleAssistant), "reponse courante"),
},
}
s.compactContext(context.Background())
if len(s.messages) != 3 {
t.Fatalf("expected 3 messages after compaction, got %d", len(s.messages))
}
if !strings.Contains(s.messages[0].Content, "resume des echanges") {
t.Errorf("first message should be the summary, got %q", s.messages[0].Content)
}
if s.messages[0].Role != string(domain.RoleUser) {
t.Errorf("summary role = %s", s.messages[0].Role)
}
if s.messages[1].Content != "question courante" || s.messages[2].Content != "reponse courante" {
t.Errorf("tail not preserved: %v", s.messages)
}
}
func TestCompactContextNoUserCut(t *testing.T) {
s := &ChatSession{
messages: []*domain.Message{
mk(string(domain.RoleUser), "seule question"),
mk(string(domain.RoleAssistant), "reponse"),
},
}
s.compactContext(context.Background())
if len(s.messages) != 2 {
t.Errorf("nothing to compact, expected 2 messages, got %d", len(s.messages))
}
}
+49 -13
View File
@@ -18,6 +18,7 @@ type Service struct {
repoChat domain.ChatRepository
llmEngine domain.LLMEngine
runningSessions map[string]*ChatSession
sessionCancels map[string]context.CancelFunc
toolService *tool.Service
mu sync.Mutex
}
@@ -38,6 +39,7 @@ func NewService(
llmEngine: llmEngine,
toolService: toolService,
runningSessions: make(map[string]*ChatSession),
sessionCancels: make(map[string]context.CancelFunc),
}
}
@@ -134,6 +136,17 @@ func (s *Service) AddChatMessage(
cb = params.OnChunk
}
contextMax := defaultContextMaxTokens
maxResponse := defaultMaxResponseTokens
if params != nil {
if params.ContextMaxTokens > 0 {
contextMax = params.ContextMaxTokens
}
if params.MaxResponseTokens > 0 {
maxResponse = params.MaxResponseTokens
}
}
s.runQuery(
ctx,
userID,
@@ -142,6 +155,8 @@ func (s *Service) AddChatMessage(
agent,
provider,
chatModelID,
contextMax,
maxResponse,
messages,
cb,
)
@@ -155,8 +170,8 @@ func (s *Service) AddChatMessage(
return message, nil
}
func (s *Service) ListChats(userID string, teamID string, page int) ([]*domain.Chat, error) {
return s.repoChat.ListChats(userID, teamID, page)
func (s *Service) ListChats(userID string, teamID string, page int, query string) ([]*domain.Chat, error) {
return s.repoChat.ListChats(userID, teamID, page, query)
}
func (s *Service) GetChat(userID string, teamID, chatID string) (*domain.Chat, error) {
@@ -180,6 +195,15 @@ func (s *Service) GetChatWithMessages(userID string, teamID, chatID string) (*do
}
func (s *Service) DeleteChat(userID string, teamID, chatID string) error {
// stoppe une éventuelle session en cours avant la suppression
s.mu.Lock()
if cancel, ok := s.sessionCancels[chatID]; ok {
cancel()
}
delete(s.runningSessions, chatID)
delete(s.sessionCancels, chatID)
s.mu.Unlock()
return s.repoChat.DeleteChat(userID, teamID, chatID)
}
@@ -191,30 +215,38 @@ func (s *Service) runQuery(
agent *domain.Agent,
provider *domain.Provider,
modelID string,
contextMaxTokens int,
maxResponseTokens int,
messages []*domain.Message,
callback domain.StreamCallback,
) error {
chatSession := &ChatSession{
userID: userID,
teamID: teamID,
chatID: chatID,
agent: agent,
provider: provider,
modelID: modelID,
messages: messages,
llmEngine: s.llmEngine,
toolService: s.toolService,
subscribers: make([]domain.StreamCallback, 0),
repoChat: s.repoChat,
userID: userID,
teamID: teamID,
chatID: chatID,
agent: agent,
provider: provider,
modelID: modelID,
messages: messages,
contextMaxTokens: contextMaxTokens,
maxResponseTokens: maxResponseTokens,
llmEngine: s.llmEngine,
toolService: s.toolService,
subscribers: make([]domain.StreamCallback, 0),
repoChat: s.repoChat,
}
if callback != nil {
chatSession.subscribers = append(chatSession.subscribers, callback)
}
// contexte annulable : permet d'arrêter la session (ex. suppression du chat)
ctx, cancel := context.WithCancel(ctx)
s.mu.Lock()
s.runningSessions[chatID] = chatSession
s.sessionCancels[chatID] = cancel
s.mu.Unlock()
go func() {
@@ -222,7 +254,11 @@ func (s *Service) runQuery(
if err != nil {
slog.Info("chat session error", "error", err)
}
s.mu.Lock()
delete(s.runningSessions, chatID)
delete(s.sessionCancels, chatID)
s.mu.Unlock()
cancel()
}()
return nil
+19 -12
View File
@@ -14,18 +14,20 @@ import (
var systemSystemPrompt string
type ChatSession struct {
userID string
teamID string
chatID string
agent *domain.Agent
provider *domain.Provider
modelID string
messages []*domain.Message
llmEngine domain.LLMEngine
toolService *tool.Service
repoChat domain.ChatRepository
subscribers []domain.StreamCallback
mu sync.Mutex
userID string
teamID string
chatID string
agent *domain.Agent
provider *domain.Provider
modelID string
messages []*domain.Message
contextMaxTokens int
maxResponseTokens int
llmEngine domain.LLMEngine
toolService *tool.Service
repoChat domain.ChatRepository
subscribers []domain.StreamCallback
mu sync.Mutex
}
func (s *ChatSession) run(ctx context.Context) error {
@@ -57,6 +59,11 @@ func (s *ChatSession) run(ctx context.Context) error {
// LOOP on TOOLS
for continue_loop {
// compaction du contexte : il doit rester assez de place pour la réponse
if s.shouldCompact(systemPrompt.Content) {
s.compactContext(ctx)
}
msg := s.llmEngine.Stream(
ctx,
s.provider,
+27 -2
View File
@@ -3,10 +3,30 @@ package chat
import (
"context"
"log/slog"
"strings"
"trankilou.fr/lassistanoque/backend/internal/domain"
)
// markdownDecorations liste les marqueurs markdown susceptibles d'apparaître
// dans un titre généré par un LLM.
var markdownDecorations = []string{"**", "__", "~~", "*", "_", "~", "#", "`"}
func sanitizeTitle(title string) string {
title = strings.TrimSpace(title)
// le titre doit tenir sur une ligne
title = strings.ReplaceAll(title, "\n", " ")
// guillemets et tirets d'encadrement ajoutés par le LLM
title = strings.Trim(title, `"'«»`)
// décorations markdown
for _, marker := range markdownDecorations {
title = strings.ReplaceAll(title, marker, "")
}
// espaces multiples créés par les suppressions
title = strings.Join(strings.Fields(title), " ")
return strings.TrimSpace(title)
}
func (s *Service) RefreshTitles(provider *domain.Provider, modelID string, teamID string) {
ctx := context.Background()
@@ -24,14 +44,19 @@ func (s *Service) RefreshTitles(provider *domain.Provider, modelID string, teamI
slog.Error("RefreshTitles.GetMessages", "error", err)
}
title, err := s.llmEngine.Generate(ctx, provider, modelID, append(messages, &domain.Message{
Role: string(domain.RoleSystem),
Content: "Tu génères un titre de conversation. Réponds uniquement par le titre, " +
"en texte brut : jamais de markdown (pas de **, *, _, ~, #), " +
"pas de guillemets, pas de liste, pas de ponctuation finale.",
}, &domain.Message{
Role: string(domain.RoleUser),
Content: "resume toute la conversation dans un titre entre 5 et 10 mots",
Content: "Résume toute la conversation dans un titre entre 5 et 10 mots, en texte brut, sans aucun formatage.",
}))
if err != nil {
slog.Error("RefreshTitles.Generate", "error", err)
}
newchat.FreshTitle = true
newchat.Title = title
newchat.Title = sanitizeTitle(title)
s.repoChat.UpdateChat(newchat.UserID, newchat)
}
}