feat(import): split sessions into per-turn units with bounded token usage · Entire
feat(import): split sessions into per-turn units with bounded token usage
d2a752c→main·
computermode·3w ago·2 files·+219 added/-0 removed
Co-Authored-By: Claude Opus 4.8 noreply@anthropic.com
Sessions
3866c6eb201dView transcript
[?
Implement Claude History Import FeatureClaude Code·Opus 4.8·3 steps](/content/gh/entireio/cli/session/897809d4-ecab-4dc6-aa21-97661cb62607#timeline-3866c6eb201d/index.html)
Changes
2
cmd/entire/cli/importclaude
Aturns.go+161
Aturns_test.go+58
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
package importclaude
import (
"crypto/sha256"
"encoding/hex"
"encoding/json"
"fmt"
"strings"
"time"
"github.com/entireio/cli/cmd/entire/cli/agent/claudecode"
"github.com/entireio/cli/cmd/entire/cli/agent/types"
"github.com/entireio/cli/cmd/entire/cli/transcript"
)
// Turn is one user-prompt turn extracted from a session transcript. Line
// offsets are in raw-line space (newline-counted), matching the offsets used by
// transcript.SliceFromLine and the agent token-usage helpers.
type Turn struct {
LineStart, LineEnd int
UUID, ParentUUID string
Prompt, Model string
CreatedAt time.Time
Tokens *types.TokenUsage
ContentHash string
}
// extraFields are line fields not modeled by transcript.Line.
type extraFields struct {
ParentUUID string `json:"parentUuid"`
Timestamp string `json:"timestamp"`
Message struct {
Model string `json:"model"`
} `json:"message"`
}
// SplitTurns produces one Turn per user-prompt line. Token usage for each turn
// is computed on the slice [LineStart, LineEnd) so turns don't double-count
// later turns. tool_result lines (Type == "user" but no text content) do not
// start a turn.
func SplitTurns(full []byte, subagentsDir string) ([]Turn, error) {
rawLines := splitRawLines(full)
// Identify user-prompt turn starts in raw-line space.
var starts []int
for i, raw := range rawLines {
if isUserPromptLine(raw) {
starts = append(starts, i)
}
}
ag := &claudecode.ClaudeCodeAgent{}
urns := make([]Turn, 0, len(starts))
for k, start := range starts {
end := len(rawLines)
if k+1 < len(starts) {
end = starts[k+1]
}
// Bound token usage to [start, end): truncate to the first `end` lines,
// then let the agent helper slice from `start`.
truncated := joinLines(rawLines[:end])
tokens, err := ag.CalculateTotalTokenUsage(truncated, start, subagentsDir)
if err != nil {
return nil, fmt.Errorf("token usage for turn %d: %w", k, err)
}
var line transcript.Line
_ = json.Unmarshal(rawLines[start], &line)
var ex extraFields
_ = json.Unmarshal(rawLines[start], &ex)
ts, _ := time.Parse(time.RFC3339, ex.Timestamp)
slice := joinLines(rawLines[start:end])
sum := sha256.Sum256(slice)
turns = append(turns, Turn{
LineStart: start,
LineEnd: end,
UUID: line.UUID,
ParentUUID: ex.ParentUUID,
Prompt: transcript.ExtractUserContent(line.Message),
Model: modelInRange(rawLines, start, end),
CreatedAt: ts,
Tokens: tokens,
ContentHash: "sha256:" + hex.EncodeToString(sum[:]),
})
}
return turns, nil
}
// modelInRange returns the model from the first assistant message within
// [start, end), or "" when none carries one. The model lives on assistant
// lines, not the user-prompt line.
func modelInRange(rawLines [][]byte, start, end int) string {
for i := start; i < end && i < len(rawLines); i++ {
var line transcript.Line
if err := json.Unmarshal(rawLines[i], &line); err != nil {
continue
}
if line.Type != "assistant" {
continue
}
var ex extraFields
if err := json.Unmarshal(rawLines[i], &ex); err != nil {
continue
}
if ex.Message.Model != "" {
return ex.Message.Model
}
}
return ""
}
// isUserPromptLine reports whether a raw JSONL line is a genuine user-prompt
// turn start: type "user" (or role "user") with non-empty extractable text.
// tool_result lines are type "user" but carry no text, so they return false.
func isUserPromptLine(raw []byte) bool {
var line transcript.Line
if err := json.Unmarshal(raw, &line); err != nil {
return false
}
typ := line.Type
if typ == "" {
typ = line.Role
}
if typ != "user" {
return false
}
return transcript.ExtractUserContent(line.Message) != ""
}
// splitRawLines splits content into raw lines in the same index space as
// transcript.SliceFromLine (newline-counted). Trailing empty segment from a
// final newline is dropped.
func splitRawLines(content []byte) [][]byte {
if len(content) == 0 {
return nil
}
parts := strings.Split(string(content), "\n")
if len(parts) > 0 && parts[len(parts)-1] == "" {
parts = parts[:len(parts)-1]
}
out := make([][]byte, len(parts))
for i, p := range parts {
out[i] = []byte(p)
}
return out
}
// joinLines reassembles raw lines into newline-terminated bytes.
func joinLines(lines [][]byte) []byte {
if len(lines) == 0 {
return nil
}
strs := make([]string, len(lines))
for i, l := range lines {
strs[i] = string(l)
}
return []byte(strings.Join(strs, "\n") + "\n")
}
Acmd/entire/cli/importclaude/turns.go+161
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
package importclaude
import (
"strings"
"testing"
)
func TestSplitTurns_TwoPromptsBoundedByNext(t *testing.T) {
t.Parallel()
full := []byte(strings.Join([]string{
`{"type":"user","uuid":"u1","parentUuid":"","timestamp":"2026-06-20T00:00:00Z","message":{"role":"user","content":"first"}}`,
`{"type":"assistant","uuid":"a1","message":{"id":"m1","model":"claude-x","content":[{"type":"text","text":"ok"}],"usage":{"input_tokens":10,"output_tokens":5}}}`,
`{"type":"user","uuid":"u2","parentUuid":"a1","timestamp":"2026-06-20T00:01:00Z","message":{"role":"user","content":"second"}}`,
`{"type":"assistant","uuid":"a2","message":{"id":"m2","model":"claude-x","content":[{"type":"text","text":"done"}],"usage":{"input_tokens":20,"output_tokens":7}}}`,
}, "\n") + "\n")
turns, err := SplitTurns(full, "")
if err != nil {
t.Fatal(err)
}
if len(turns) != 2 {
t.Fatalf("want 2 turns, got %d", len(turns))
}
if turns[0].LineStart != 0 || turns[0].LineEnd != 2 {
t.Errorf("turn0 bounds = [%d,%d), want [0,2)", turns[0].LineStart, turns[0].LineEnd)
}
if turns[0].Prompt != "first" || turns[1].Prompt != "second" {
t.Errorf("prompts = %q,%q", turns[0].Prompt, turns[1].Prompt)
}
if turns[0].UUID != "u1" || turns[1].ParentUUID != "a1" {
t.Errorf("uuid/parent wrong: %+v %+v", turns[0], turns[1])
}
if turns[0].Model != "claude-x" {
t.Errorf("turn0 model = %q, want claude-x", turns[0].Model)
}
if turns[0].Tokens == nil || turns[0].Tokens.OutputTokens != 5 {
t.Errorf("turn0 tokens not bounded to its own turn: %+v", turns[0].Tokens)
}
if turns[1].Tokens == nil || turns[1].Tokens.OutputTokens != 7 {
t.Errorf("turn1 tokens wrong: %+v", turns[1].Tokens)
}
}
func TestSplitTurns_ToolResultIsNotATurn(t *testing.T) {
t.Parallel()
full := []byte(strings.Join([]string{
`{"type":"user","uuid":"u1","message":{"role":"user","content":"do it"}}`,
`{"type":"assistant","uuid":"a1","message":{"id":"m1","content":[{"type":"tool_use","id":"t1","name":"Bash","input":{}}],"usage":{"output_tokens":3}}}`,
`{"type":"user","uuid":"r1","message":{"content":[{"type":"tool_result","tool_use_id":"t1","content":"out"}]}}`,
}, "\n") + "\n")
turns, err := SplitTurns(full, "")
if err != nil {
t.Fatal(err)
}
if len(turns) != 1 {
t.Fatalf("tool_result must not start a turn; want 1 turn, got %d", len(turns))
}
}