compaction de contexte, filtre/pagination/suppression des conversations
- compaction automatique du contexte LLM quand la taille atteint le seuil (contexte max - taille de réponse max), par résumé de l'historique ancien - filtre textuel (?q=) et pagination (20/page) de l'historique des conversations - correctif ORM : offset de pagination en base 0 - suppression d'une conversation : messages associés supprimés, session en cours annulée, popup de confirmation (ConfirmDialog) - titres de conversation nettoyés du markdown - styles des listes markdown, variante danger du bouton - go.mod/go.sum : dépendances go-rod et stealth manquantes du commit précédent
This commit is contained in:
1 parent
a970bfa991
commit
51bbb45b32
86 files changed
+770
-113
No files matched your search
@@ -0,0 +1,105 @@
|
||||
package chat
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"strings"
|
||||
|
||||
"trankilou.fr/lassistanoque/backend/internal/domain"
|
||||
)
|
||||
|
||||
const (
|
||||
// charsPerToken : estimation heuristique, ~4 caractères par token pour du
|
||||
// texte européen. Suffisant pour décider d'une compaction.
|
||||
charsPerToken = 4
|
||||
|
||||
// defaultContextMaxTokens : taille de contexte du modèle quand la
|
||||
// configuration ne la précise pas.
|
||||
defaultContextMaxTokens = 32000
|
||||
|
||||
// defaultMaxResponseTokens : réserve à ménager pour la réponse du modèle.
|
||||
defaultMaxResponseTokens = 4096
|
||||
)
|
||||
|
||||
// estimateTokens estime le nombre de tokens d'un texte.
|
||||
func estimateTokens(text string) int {
|
||||
if text == "" {
|
||||
return 0
|
||||
}
|
||||
tokens := len(text) / charsPerToken
|
||||
if len(text)%charsPerToken > 0 {
|
||||
tokens++
|
||||
}
|
||||
return tokens
|
||||
}
|
||||
|
||||
// contextTokens estime la taille en tokens du prompt complet envoyé au LLM.
|
||||
func (s *ChatSession) contextTokens(systemPrompt string) int {
|
||||
total := estimateTokens(systemPrompt)
|
||||
for _, m := range s.messages {
|
||||
total += estimateTokens(m.Content) + estimateTokens(m.ToolCallsJson)
|
||||
}
|
||||
return total
|
||||
}
|
||||
|
||||
// shouldCompact indique si la taille du contexte a atteint la limite au-delà
|
||||
// de laquelle il ne reste plus assez de place pour une réponse complète :
|
||||
// contexte max diminué de la taille de réponse max.
|
||||
func (s *ChatSession) shouldCompact(systemPrompt string) bool {
|
||||
return s.contextTokens(systemPrompt) >= s.contextMaxTokens-s.maxResponseTokens
|
||||
}
|
||||
|
||||
// compactContext remplace l'historique ancien par un résumé généré par le LLM.
|
||||
// Les échanges récents (à partir du dernier message utilisateur) sont conservés
|
||||
// intacts : un point de coupure sur un message utilisateur ne peut jamais
|
||||
// séparer un message assistant porteur d'appels d'outils de ses réponses tool,
|
||||
// ce qui garderait un historique incohérent pour l'API.
|
||||
// La compaction ne porte que sur la session en mémoire : l'historique complet
|
||||
// reste intact en base.
|
||||
func (s *ChatSession) compactContext(ctx context.Context) {
|
||||
cut := -1
|
||||
for i := len(s.messages) - 1; i >= 0; i-- {
|
||||
if s.messages[i].Role == string(domain.RoleUser) {
|
||||
cut = i
|
||||
break
|
||||
}
|
||||
}
|
||||
// rien à compacter : l'historique commence déjà au dernier message utilisateur
|
||||
if cut <= 0 {
|
||||
return
|
||||
}
|
||||
|
||||
head := s.messages[:cut]
|
||||
tail := s.messages[cut:]
|
||||
|
||||
var sb strings.Builder
|
||||
for _, m := range head {
|
||||
fmt.Fprintf(&sb, "%s: %s\n", m.Role, m.Content)
|
||||
}
|
||||
|
||||
summary, err := s.llmEngine.Generate(ctx, s.provider, s.modelID, []*domain.Message{
|
||||
{
|
||||
Role: string(domain.RoleUser),
|
||||
Content: "Résume de façon concise et factuelle la conversation suivante. " +
|
||||
"Conserve les informations essentielles : demandes, décisions, résultats, fichiers ou liens mentionnés. " +
|
||||
"Réponds en texte brut, sans markdown.\n\n" + sb.String(),
|
||||
},
|
||||
})
|
||||
if err != nil {
|
||||
slog.Error("compactContext.Generate", "error", err)
|
||||
return
|
||||
}
|
||||
|
||||
summaryMessage := &domain.Message{
|
||||
Role: string(domain.RoleUser),
|
||||
Content: "Résumé de la conversation précédente (les messages plus anciens ont été compactés) :\n" +
|
||||
summary,
|
||||
}
|
||||
|
||||
s.messages = append([]*domain.Message{summaryMessage}, tail...)
|
||||
slog.Info("context compacted",
|
||||
"removedMessages", len(head),
|
||||
"keptMessages", len(tail)+1,
|
||||
)
|
||||
}
|
||||
@@ -0,0 +1,97 @@
|
||||
package chat
|
||||
|
||||
import (
|
||||
"context"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"trankilou.fr/lassistanoque/backend/internal/domain"
|
||||
)
|
||||
|
||||
type fakeEngine struct{ summary string }
|
||||
|
||||
func (f *fakeEngine) ListProviderTypes() []domain.Item { return nil }
|
||||
func (f *fakeEngine) ListModelsFromProvider(ctx context.Context, p *domain.Provider) ([]string, error) {
|
||||
return nil, nil
|
||||
}
|
||||
func (f *fakeEngine) Stream(ctx context.Context, p *domain.Provider, m string, params *domain.LLMParams, msgs []*domain.Message) *domain.Message {
|
||||
return &domain.Message{}
|
||||
}
|
||||
func (f *fakeEngine) Generate(ctx context.Context, p *domain.Provider, m string, msgs []*domain.Message) (string, error) {
|
||||
return f.summary, nil
|
||||
}
|
||||
|
||||
func mk(role, content string) *domain.Message {
|
||||
return &domain.Message{Role: role, Content: content}
|
||||
}
|
||||
|
||||
func TestEstimateTokens(t *testing.T) {
|
||||
if got := estimateTokens(""); got != 0 {
|
||||
t.Errorf("empty = %d", got)
|
||||
}
|
||||
if got := estimateTokens("abcd"); got != 1 {
|
||||
t.Errorf("abcd = %d", got)
|
||||
}
|
||||
if got := estimateTokens("abcde"); got != 2 {
|
||||
t.Errorf("abcde = %d", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestShouldCompactThreshold(t *testing.T) {
|
||||
s := &ChatSession{
|
||||
messages: []*domain.Message{mk(string(domain.RoleUser), strings.Repeat("a", 320))}, // 80 tokens
|
||||
contextMaxTokens: 100,
|
||||
maxResponseTokens: 20,
|
||||
}
|
||||
if !s.shouldCompact("system") { // 80 + 1 >= 100-20
|
||||
t.Errorf("should compact at contextMax - maxResponse boundary")
|
||||
}
|
||||
s.maxResponseTokens = 40
|
||||
s.contextMaxTokens = 130 // seuil 130-40=90 > 82 : pas de compaction
|
||||
if s.shouldCompact("system") {
|
||||
t.Errorf("should not compact below threshold")
|
||||
}
|
||||
}
|
||||
|
||||
func TestCompactContext(t *testing.T) {
|
||||
s := &ChatSession{
|
||||
provider: &domain.Provider{},
|
||||
llmEngine: &fakeEngine{summary: "resume des echanges"},
|
||||
contextMaxTokens: 100,
|
||||
maxResponseTokens: 10,
|
||||
messages: []*domain.Message{
|
||||
mk(string(domain.RoleUser), "ancienne question 1"),
|
||||
mk(string(domain.RoleAssistant), "ancienne reponse 1"),
|
||||
mk(string(domain.RoleUser), "question courante"),
|
||||
mk(string(domain.RoleAssistant), "reponse courante"),
|
||||
},
|
||||
}
|
||||
|
||||
s.compactContext(context.Background())
|
||||
|
||||
if len(s.messages) != 3 {
|
||||
t.Fatalf("expected 3 messages after compaction, got %d", len(s.messages))
|
||||
}
|
||||
if !strings.Contains(s.messages[0].Content, "resume des echanges") {
|
||||
t.Errorf("first message should be the summary, got %q", s.messages[0].Content)
|
||||
}
|
||||
if s.messages[0].Role != string(domain.RoleUser) {
|
||||
t.Errorf("summary role = %s", s.messages[0].Role)
|
||||
}
|
||||
if s.messages[1].Content != "question courante" || s.messages[2].Content != "reponse courante" {
|
||||
t.Errorf("tail not preserved: %v", s.messages)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCompactContextNoUserCut(t *testing.T) {
|
||||
s := &ChatSession{
|
||||
messages: []*domain.Message{
|
||||
mk(string(domain.RoleUser), "seule question"),
|
||||
mk(string(domain.RoleAssistant), "reponse"),
|
||||
},
|
||||
}
|
||||
s.compactContext(context.Background())
|
||||
if len(s.messages) != 2 {
|
||||
t.Errorf("nothing to compact, expected 2 messages, got %d", len(s.messages))
|
||||
}
|
||||
}
|
||||
@@ -18,6 +18,7 @@ type Service struct {
|
||||
repoChat domain.ChatRepository
|
||||
llmEngine domain.LLMEngine
|
||||
runningSessions map[string]*ChatSession
|
||||
sessionCancels map[string]context.CancelFunc
|
||||
toolService *tool.Service
|
||||
mu sync.Mutex
|
||||
}
|
||||
@@ -38,6 +39,7 @@ func NewService(
|
||||
llmEngine: llmEngine,
|
||||
toolService: toolService,
|
||||
runningSessions: make(map[string]*ChatSession),
|
||||
sessionCancels: make(map[string]context.CancelFunc),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -134,6 +136,17 @@ func (s *Service) AddChatMessage(
|
||||
cb = params.OnChunk
|
||||
}
|
||||
|
||||
contextMax := defaultContextMaxTokens
|
||||
maxResponse := defaultMaxResponseTokens
|
||||
if params != nil {
|
||||
if params.ContextMaxTokens > 0 {
|
||||
contextMax = params.ContextMaxTokens
|
||||
}
|
||||
if params.MaxResponseTokens > 0 {
|
||||
maxResponse = params.MaxResponseTokens
|
||||
}
|
||||
}
|
||||
|
||||
s.runQuery(
|
||||
ctx,
|
||||
userID,
|
||||
@@ -142,6 +155,8 @@ func (s *Service) AddChatMessage(
|
||||
agent,
|
||||
provider,
|
||||
chatModelID,
|
||||
contextMax,
|
||||
maxResponse,
|
||||
messages,
|
||||
cb,
|
||||
)
|
||||
@@ -155,8 +170,8 @@ func (s *Service) AddChatMessage(
|
||||
return message, nil
|
||||
}
|
||||
|
||||
func (s *Service) ListChats(userID string, teamID string, page int) ([]*domain.Chat, error) {
|
||||
return s.repoChat.ListChats(userID, teamID, page)
|
||||
func (s *Service) ListChats(userID string, teamID string, page int, query string) ([]*domain.Chat, error) {
|
||||
return s.repoChat.ListChats(userID, teamID, page, query)
|
||||
}
|
||||
|
||||
func (s *Service) GetChat(userID string, teamID, chatID string) (*domain.Chat, error) {
|
||||
@@ -180,6 +195,15 @@ func (s *Service) GetChatWithMessages(userID string, teamID, chatID string) (*do
|
||||
}
|
||||
|
||||
func (s *Service) DeleteChat(userID string, teamID, chatID string) error {
|
||||
// stoppe une éventuelle session en cours avant la suppression
|
||||
s.mu.Lock()
|
||||
if cancel, ok := s.sessionCancels[chatID]; ok {
|
||||
cancel()
|
||||
}
|
||||
delete(s.runningSessions, chatID)
|
||||
delete(s.sessionCancels, chatID)
|
||||
s.mu.Unlock()
|
||||
|
||||
return s.repoChat.DeleteChat(userID, teamID, chatID)
|
||||
}
|
||||
|
||||
@@ -191,30 +215,38 @@ func (s *Service) runQuery(
|
||||
agent *domain.Agent,
|
||||
provider *domain.Provider,
|
||||
modelID string,
|
||||
contextMaxTokens int,
|
||||
maxResponseTokens int,
|
||||
messages []*domain.Message,
|
||||
callback domain.StreamCallback,
|
||||
) error {
|
||||
|
||||
chatSession := &ChatSession{
|
||||
userID: userID,
|
||||
teamID: teamID,
|
||||
chatID: chatID,
|
||||
agent: agent,
|
||||
provider: provider,
|
||||
modelID: modelID,
|
||||
messages: messages,
|
||||
llmEngine: s.llmEngine,
|
||||
toolService: s.toolService,
|
||||
subscribers: make([]domain.StreamCallback, 0),
|
||||
repoChat: s.repoChat,
|
||||
userID: userID,
|
||||
teamID: teamID,
|
||||
chatID: chatID,
|
||||
agent: agent,
|
||||
provider: provider,
|
||||
modelID: modelID,
|
||||
messages: messages,
|
||||
contextMaxTokens: contextMaxTokens,
|
||||
maxResponseTokens: maxResponseTokens,
|
||||
llmEngine: s.llmEngine,
|
||||
toolService: s.toolService,
|
||||
subscribers: make([]domain.StreamCallback, 0),
|
||||
repoChat: s.repoChat,
|
||||
}
|
||||
|
||||
if callback != nil {
|
||||
chatSession.subscribers = append(chatSession.subscribers, callback)
|
||||
}
|
||||
|
||||
// contexte annulable : permet d'arrêter la session (ex. suppression du chat)
|
||||
ctx, cancel := context.WithCancel(ctx)
|
||||
|
||||
s.mu.Lock()
|
||||
s.runningSessions[chatID] = chatSession
|
||||
s.sessionCancels[chatID] = cancel
|
||||
s.mu.Unlock()
|
||||
|
||||
go func() {
|
||||
@@ -222,7 +254,11 @@ func (s *Service) runQuery(
|
||||
if err != nil {
|
||||
slog.Info("chat session error", "error", err)
|
||||
}
|
||||
s.mu.Lock()
|
||||
delete(s.runningSessions, chatID)
|
||||
delete(s.sessionCancels, chatID)
|
||||
s.mu.Unlock()
|
||||
cancel()
|
||||
}()
|
||||
|
||||
return nil
|
||||
|
||||
@@ -14,18 +14,20 @@ import (
|
||||
var systemSystemPrompt string
|
||||
|
||||
type ChatSession struct {
|
||||
userID string
|
||||
teamID string
|
||||
chatID string
|
||||
agent *domain.Agent
|
||||
provider *domain.Provider
|
||||
modelID string
|
||||
messages []*domain.Message
|
||||
llmEngine domain.LLMEngine
|
||||
toolService *tool.Service
|
||||
repoChat domain.ChatRepository
|
||||
subscribers []domain.StreamCallback
|
||||
mu sync.Mutex
|
||||
userID string
|
||||
teamID string
|
||||
chatID string
|
||||
agent *domain.Agent
|
||||
provider *domain.Provider
|
||||
modelID string
|
||||
messages []*domain.Message
|
||||
contextMaxTokens int
|
||||
maxResponseTokens int
|
||||
llmEngine domain.LLMEngine
|
||||
toolService *tool.Service
|
||||
repoChat domain.ChatRepository
|
||||
subscribers []domain.StreamCallback
|
||||
mu sync.Mutex
|
||||
}
|
||||
|
||||
func (s *ChatSession) run(ctx context.Context) error {
|
||||
@@ -57,6 +59,11 @@ func (s *ChatSession) run(ctx context.Context) error {
|
||||
|
||||
// LOOP on TOOLS
|
||||
for continue_loop {
|
||||
// compaction du contexte : il doit rester assez de place pour la réponse
|
||||
if s.shouldCompact(systemPrompt.Content) {
|
||||
s.compactContext(ctx)
|
||||
}
|
||||
|
||||
msg := s.llmEngine.Stream(
|
||||
ctx,
|
||||
s.provider,
|
||||
|
||||
@@ -3,10 +3,30 @@ package chat
|
||||
import (
|
||||
"context"
|
||||
"log/slog"
|
||||
"strings"
|
||||
|
||||
"trankilou.fr/lassistanoque/backend/internal/domain"
|
||||
)
|
||||
|
||||
// markdownDecorations liste les marqueurs markdown susceptibles d'apparaître
|
||||
// dans un titre généré par un LLM.
|
||||
var markdownDecorations = []string{"**", "__", "~~", "*", "_", "~", "#", "`"}
|
||||
|
||||
func sanitizeTitle(title string) string {
|
||||
title = strings.TrimSpace(title)
|
||||
// le titre doit tenir sur une ligne
|
||||
title = strings.ReplaceAll(title, "\n", " ")
|
||||
// guillemets et tirets d'encadrement ajoutés par le LLM
|
||||
title = strings.Trim(title, `"'«»`)
|
||||
// décorations markdown
|
||||
for _, marker := range markdownDecorations {
|
||||
title = strings.ReplaceAll(title, marker, "")
|
||||
}
|
||||
// espaces multiples créés par les suppressions
|
||||
title = strings.Join(strings.Fields(title), " ")
|
||||
return strings.TrimSpace(title)
|
||||
}
|
||||
|
||||
func (s *Service) RefreshTitles(provider *domain.Provider, modelID string, teamID string) {
|
||||
|
||||
ctx := context.Background()
|
||||
@@ -24,14 +44,19 @@ func (s *Service) RefreshTitles(provider *domain.Provider, modelID string, teamI
|
||||
slog.Error("RefreshTitles.GetMessages", "error", err)
|
||||
}
|
||||
title, err := s.llmEngine.Generate(ctx, provider, modelID, append(messages, &domain.Message{
|
||||
Role: string(domain.RoleSystem),
|
||||
Content: "Tu génères un titre de conversation. Réponds uniquement par le titre, " +
|
||||
"en texte brut : jamais de markdown (pas de **, *, _, ~, #), " +
|
||||
"pas de guillemets, pas de liste, pas de ponctuation finale.",
|
||||
}, &domain.Message{
|
||||
Role: string(domain.RoleUser),
|
||||
Content: "resume toute la conversation dans un titre entre 5 et 10 mots",
|
||||
Content: "Résume toute la conversation dans un titre entre 5 et 10 mots, en texte brut, sans aucun formatage.",
|
||||
}))
|
||||
if err != nil {
|
||||
slog.Error("RefreshTitles.Generate", "error", err)
|
||||
}
|
||||
newchat.FreshTitle = true
|
||||
newchat.Title = title
|
||||
newchat.Title = sanitizeTitle(title)
|
||||
s.repoChat.UpdateChat(newchat.UserID, newchat)
|
||||
}
|
||||
}
|
||||
|
||||
Reference in new issue
Block a user