diff --git a/.env.example b/.env.example index d2a304a2cb1..3d587950213 100644 --- a/.env.example +++ b/.env.example @@ -354,6 +354,12 @@ OLLAMA_BASE_URL=http://host.docker.internal:11434 # BATCH_EMBED_SIZE= # VLM 单次 HTTP 请求超时(秒,默认 180;慢端点超时报 context deadline 时调大)。 # VLM_HTTP_TIMEOUT_SECONDS=180 +# EXIF 方向归一化(默认开启)。开启时按 JPEG 的 EXIF Orientation 把像素转正后 +# 再送模型(模型只读像素、不看 EXIF,横放的扫描件否则会读错)。但 EXIF 标签可能 +# 在拍摄端就写错,且两种错法在字节上无法区分:tag 说 1 而内容倒了(转不了), +# 或 tag 说 6 而内容本来就正(会被转歪)。若你的来源里出现过后者,设为 off, +# 图片将按存储原样逐字节转发。 +# VLM_IMAGE_ORIENTATION=off # LLM 流原始转储(排查上下文问题;1 启用,默认目录 ~/.weknora/investigate/llm-stream)。 # WEKNORA_LLM_STREAM_RAW_DUMP= # WEKNORA_LLM_STREAM_RAW_DUMP_DIR= diff --git a/docker-compose.yml b/docker-compose.yml index 68e3ba00d0a..548c075ec03 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -273,6 +273,8 @@ services: - BATCH_EMBED_SIZE=${BATCH_EMBED_SIZE:-} # VLM 单次 HTTP 请求超时(秒,默认 180) - VLM_HTTP_TIMEOUT_SECONDS=${VLM_HTTP_TIMEOUT_SECONDS:-} + # EXIF 方向归一化,默认开启;设为 off 则按存储原样转发图片 + - VLM_IMAGE_ORIENTATION=${VLM_IMAGE_ORIENTATION:-} # LLM 流原始转储(排查上下文问题;1 启用,默认目录 ~/.weknora/investigate/llm-stream) - WEKNORA_LLM_STREAM_RAW_DUMP=${WEKNORA_LLM_STREAM_RAW_DUMP:-} - WEKNORA_LLM_STREAM_RAW_DUMP_DIR=${WEKNORA_LLM_STREAM_RAW_DUMP_DIR:-} diff --git a/internal/models/vlm/image_orientation.go b/internal/models/vlm/image_orientation.go new file mode 100644 index 00000000000..83da5283d22 --- /dev/null +++ b/internal/models/vlm/image_orientation.go @@ -0,0 +1,327 @@ +package vlm + +import ( + "bytes" + "context" + "encoding/binary" + "image" + "image/jpeg" + "os" + "strings" + "sync" + + "github.com/Tencent/WeKnora/internal/logger" +) + +const ( + // exifOrientationTag is TIFF tag 0x0112 (Orientation). + exifOrientationTag = 0x0112 + // jpegMarkerAPP1 carries the EXIF payload. + jpegMarkerAPP1 = 0xE1 + jpegMarkerSOS = 0xDA + jpegMarkerEOI = 0xD9 + // orientationJPEGQuality re-encodes a rotated page. Only images that + // actually carry a rotation tag are re-encoded; the rest keep their bytes. + orientationJPEGQuality = 90 +) + +// maxOrientationPixels bounds the canvases this decorator is willing to decode. +// Rotation needs the whole image in memory (about four bytes per pixel), so an +// absurd canvas is left untouched rather than risking the worker. A 600 DPI A3 +// scan (about 70 megapixels) still fits. +var maxOrientationPixels = 80 << 20 + +// imageOrientationEnvVar turns the EXIF orientation pass OFF. +// +// The pass is ON by default: a page stored sideways is the common case this +// decorator exists for, and models read the pixel matrix while ignoring the tag, +// so without it such a page reaches the model rotated. But the tag is the only +// signal the pass has, and a camera can write it wrong at capture time — in both +// directions: +// +// - tag says 1 but the pixels are upside down: no rotation happens, and the +// stored (wrong) orientation reaches the model, exactly as before. +// - tag says 6 but the pixels are already upright (a landscape-held phone +// shot, for instance): the tag is honoured and CORRECT pixels are turned +// sideways, which makes that image unreadable — a regression the pass +// introduces rather than a pre-existing bug. +// +// Those two inputs are indistinguishable from the bytes alone, and asking the +// model to reason about orientation instead is not a reliable substitute +// (measured: a prompt that says "straighten the text first" recovers table +// structure on a 180-degree image but misreads individual glyphs on a +// 90-degree one). A deployment whose sources are known to carry trustworthy +// tags therefore keeps the default; one that has seen the second failure mode +// sets VLM_IMAGE_ORIENTATION=off and gets the bytes forwarded exactly as +// stored, byte for byte. +func imageOrientationEnabled() bool { + switch strings.ToLower(strings.TrimSpace(os.Getenv("VLM_IMAGE_ORIENTATION"))) { + case "0", "false", "no", "off": + return false + default: + return true + } +} + +// maxConcurrentOrientationConversions caps how many EXIF conversions this +// process runs at the same time. It covers every VLM instance and both the +// interactive and the background path. +// +// One conversion holds three full-size buffers at once: the decoded image +// (about 1.5 bytes per pixel for a JPEG), the RGBA canvas it is turned onto +// (4 bytes per pixel) and the re-encoded JPEG. At the pixel budget above that +// is on the order of 0.5 GiB of transient memory, so a single slot is what +// bounds the total: however many models, uploads or workers are in flight, the +// process never holds more than one conversion's worth of it. The limit is the +// only knob here; raise it if conversion throughput ever outweighs that. +const maxConcurrentOrientationConversions = 1 + +// orientationSlots is the process-wide conversion budget. It is deliberately +// package-level rather than a field of orientationVLM: a process has one memory +// budget, while VLM instances are created per model and re-created on every +// configuration change, so a per-instance budget would multiply the bound by +// the number of live instances — the exact stacking this gate exists to stop. +var orientationSlots = make(chan struct{}, maxConcurrentOrientationConversions) + +// acquireOrientationSlot takes one conversion slot without waiting. ok is false +// when the budget is exhausted, and the caller then sends the image as stored, +// exactly as it does for a canvas above maxOrientationPixels. Waiting is +// deliberately not an option: the gate also sits on the interactive chat path, +// which must not queue behind someone else's scan when it can simply forward +// the bytes it was given. +func acquireOrientationSlot() (release func(), ok bool) { + select { + case orientationSlots <- struct{}{}: + var once sync.Once + return func() { once.Do(func() { <-orientationSlots }) }, true + default: + return nil, false + } +} + +// decodeImage is image.Decode behind a variable so the concurrency tests can +// keep a conversion inside the decoder — the only point where a slot stays held +// long enough for another call to observe — and count how often the decoder was +// reached. +var decodeImage = image.Decode + +// orientationVLM applies the EXIF Orientation tag to the pixels before an image +// reaches a model. +// +// Cameras and scanners store the sensor's pixel matrix as captured and record +// how it must be turned for display in EXIF tag 0x0112. Browsers and image +// viewers honour that tag, so an upload looks upright everywhere in the UI, but +// vision models read the pixel matrix and generally ignore the tag: a portrait +// page stored sideways (orientation 6/8) or upside down (3) reaches the model +// rotated. The model then has to recognise rotated glyphs and run table layout +// reasoning in a rotated frame, which is where merged-cell anchors, names and +// numbers start drifting. +type orientationVLM struct { + inner VLM +} + +func (w *orientationVLM) GetModelName() string { return w.inner.GetModelName() } +func (w *orientationVLM) GetModelID() string { return w.inner.GetModelID() } + +func (w *orientationVLM) Predict(ctx context.Context, imgBytes [][]byte, prompt string) (string, error) { + if len(imgBytes) == 0 { + return w.inner.Predict(ctx, imgBytes, prompt) + } + oriented := make([][]byte, len(imgBytes)) + for i, img := range imgBytes { + oriented[i] = orientImageBytes(ctx, img) + } + return w.inner.Predict(ctx, oriented, prompt) +} + +// wrapVLMImageOrientation installs the EXIF orientation normaliser as the +// innermost VLM decorator, so every implementation (remote API, Ollama, cloud) +// and every caller sees upright pixels while the debug and Langfuse layers +// record what was really sent. +func wrapVLMImageOrientation(v VLM, err error) (VLM, error) { + if err != nil || v == nil { + return v, err + } + return &orientationVLM{inner: v}, nil +} + +// orientImageBytes returns image bytes whose pixels match their EXIF +// orientation. Images without a rotation tag, non-JPEG payloads, unreadable +// files, canvases above maxOrientationPixels and calls that find every +// conversion slot busy are returned unchanged — the decorator never blocks a +// call it cannot improve. +// +// The whole pass is behind imageOrientationEnabled, which is off unless the +// deployment asks for it. See the rationale there. +func orientImageBytes(ctx context.Context, data []byte) []byte { + if !imageOrientationEnabled() { + return data + } + orientation := jpegEXIFOrientation(data) + if orientation <= 1 || orientation > 8 { + return data + } + cfg, _, err := image.DecodeConfig(bytes.NewReader(data)) + if err != nil || cfg.Width <= 0 || cfg.Height <= 0 { + return data + } + if cfg.Width*cfg.Height > maxOrientationPixels { + logger.Warnf(ctx, + "[VLM] Image %dx%d exceeds the %d pixel rotation budget; sending it in its stored orientation", + cfg.Width, cfg.Height, maxOrientationPixels) + return data + } + // The header probe above allocates nothing; the slot is taken for the + // expensive window only (decode -> turn -> re-encode) and held until that + // window closes, on every return path below. + release, acquired := acquireOrientationSlot() + if !acquired { + logger.Warnf(ctx, + "[VLM] All %d rotation slots are busy; sending the %dx%d image in its stored orientation", + maxConcurrentOrientationConversions, cfg.Width, cfg.Height) + return data + } + defer release() + img, _, err := decodeImage(bytes.NewReader(data)) + if err != nil { + return data + } + rotated := applyEXIFOrientation(img, orientation) + if rotated == nil { + return data + } + var buf bytes.Buffer + if err := jpeg.Encode(&buf, rotated, &jpeg.Options{Quality: orientationJPEGQuality}); err != nil { + return data + } + logger.Infof(ctx, "[VLM] Applied EXIF orientation %d: %dx%d -> %dx%d, %d -> %d bytes", + orientation, cfg.Width, cfg.Height, rotated.Bounds().Dx(), rotated.Bounds().Dy(), len(data), buf.Len()) + return buf.Bytes() +} + +// applyEXIFOrientation rebuilds the image so that no rotation or mirroring is +// left to the viewer. The mapping is the one EXIF defines for each value: +// 2 mirrors horizontally, 4 vertically, 3 turns 180 degrees, 6 turns 90 degrees +// clockwise, 8 turns 90 degrees counter-clockwise, and 5/7 are the transposed +// (mirrored diagonal) variants. Values outside 2..8 return nil. +func applyEXIFOrientation(src image.Image, orientation int) image.Image { + bounds := src.Bounds() + width, height := bounds.Dx(), bounds.Dy() + if width == 0 || height == 0 { + return nil + } + dstWidth, dstHeight := width, height + if orientation >= 5 { + dstWidth, dstHeight = height, width + } + dst := image.NewRGBA(image.Rect(0, 0, dstWidth, dstHeight)) + for y := range dstHeight { + for x := range dstWidth { + var sx, sy int + switch orientation { + case 2: + sx, sy = width-1-x, y + case 3: + sx, sy = width-1-x, height-1-y + case 4: + sx, sy = x, height-1-y + case 5: + sx, sy = y, x + case 6: + sx, sy = y, height-1-x + case 7: + sx, sy = width-1-y, height-1-x + case 8: + sx, sy = width-1-y, x + default: + return nil + } + dst.Set(x, y, src.At(bounds.Min.X+sx, bounds.Min.Y+sy)) + } + } + return dst +} + +// jpegEXIFOrientation returns the EXIF Orientation of a JPEG, or 0 when the +// payload is not a JPEG, carries no EXIF block, or the tag cannot be read +// without guessing. Only segment markers are walked — image data is never +// touched — so the cost is proportional to the headers, not the file. +func jpegEXIFOrientation(data []byte) int { + if len(data) < 4 || data[0] != 0xFF || data[1] != 0xD8 { + return 0 + } + for pos := 2; pos+4 <= len(data); { + if data[pos] != 0xFF { + return 0 // Desynchronised: refuse to interpret the rest. + } + marker := data[pos+1] + switch { + case marker == jpegMarkerSOS || marker == jpegMarkerEOI: + return 0 + case marker == 0x00 || marker == 0xFF: + pos++ // Padding or a stuffed byte. + continue + case marker >= 0xD0 && marker <= 0xD8: + pos += 2 // Standalone marker without a payload. + continue + } + segLen := int(binary.BigEndian.Uint16(data[pos+2 : pos+4])) + if segLen < 2 || pos+2+segLen > len(data) { + return 0 + } + if marker == jpegMarkerAPP1 { + if orientation := exifTIFFOrientation(data[pos+4 : pos+2+segLen]); orientation != 0 { + return orientation + } + } + pos += 2 + segLen + } + return 0 +} + +// exifTIFFOrientation reads tag 0x0112 out of an APP1 payload ("Exif\0\0" +// followed by a TIFF header and IFD0). +func exifTIFFOrientation(segment []byte) int { + const exifHeader = "Exif\x00\x00" + if len(segment) < len(exifHeader)+8 || string(segment[:len(exifHeader)]) != exifHeader { + return 0 + } + tiff := segment[len(exifHeader):] + var order binary.ByteOrder + switch { + case tiff[0] == 'I' && tiff[1] == 'I': + order = binary.LittleEndian + case tiff[0] == 'M' && tiff[1] == 'M': + order = binary.BigEndian + default: + return 0 + } + if order.Uint16(tiff[2:4]) != 42 { + return 0 + } + offset := int(order.Uint32(tiff[4:8])) + if offset < 8 || offset+2 > len(tiff) { + return 0 + } + entries := int(order.Uint16(tiff[offset : offset+2])) + for i := range entries { + entry := offset + 2 + i*12 + if entry+12 > len(tiff) { + return 0 + } + if order.Uint16(tiff[entry:entry+2]) != exifOrientationTag { + continue + } + // Orientation is a SHORT, so its value sits in the first two bytes of + // the entry's value field. + if order.Uint16(tiff[entry+2:entry+4]) != 3 { + return 0 + } + value := int(order.Uint16(tiff[entry+8 : entry+10])) + if value >= 1 && value <= 8 { + return value + } + return 0 + } + return 0 +} diff --git a/internal/models/vlm/image_orientation_concurrency_test.go b/internal/models/vlm/image_orientation_concurrency_test.go new file mode 100644 index 00000000000..c6c02d649f7 --- /dev/null +++ b/internal/models/vlm/image_orientation_concurrency_test.go @@ -0,0 +1,287 @@ +package vlm + +import ( + "bytes" + "context" + "encoding/binary" + "image" + "io" + "sync" + "sync/atomic" + "testing" + "time" + + "github.com/Tencent/WeKnora/internal/types" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// orientationDecodeProbe replaces the decoder seam for one test. It counts +// every call; when hold is set the FIRST call stops inside the decoder and +// reports it on entered, so a test can keep the (single) conversion slot busy +// for exactly as long as it needs instead of racing a timer. Later calls go +// straight to the real decoder, so a test whose gate is missing fails on an +// assertion instead of deadlocking on a conversion nobody releases. +type orientationDecodeProbe struct { + calls atomic.Int64 + hold bool + entered chan struct{} + release chan struct{} + once sync.Once +} + +func installOrientationDecode(t *testing.T, probe *orientationDecodeProbe) *orientationDecodeProbe { + t.Helper() + previous := decodeImage + decodeImage = func(r io.Reader) (image.Image, string, error) { + if probe.calls.Add(1) == 1 && probe.hold { + probe.entered <- struct{}{} + <-probe.release + } + return previous(r) + } + t.Cleanup(func() { + probe.releaseAll() + decodeImage = previous + }) + return probe +} + +// holdOrientationDecode stops the first conversion inside the decoder, which is +// how a test occupies the only conversion slot deterministically. +func holdOrientationDecode(t *testing.T) *orientationDecodeProbe { + t.Helper() + return installOrientationDecode(t, &orientationDecodeProbe{ + hold: true, + entered: make(chan struct{}, 1), + release: make(chan struct{}), + }) +} + +// countOrientationDecodes only counts, so a conversion that unexpectedly +// reaches the decoder fails an assertion instead of blocking. +func countOrientationDecodes(t *testing.T) *orientationDecodeProbe { + t.Helper() + return installOrientationDecode(t, &orientationDecodeProbe{ + entered: make(chan struct{}, 1), + release: make(chan struct{}), + }) +} + +// waitInsideDecoder blocks until the held conversion has reached the decoder. +// The timeout guards a broken fixture; it never decides the outcome of the test. +func (p *orientationDecodeProbe) waitInsideDecoder(t *testing.T) { + t.Helper() + select { + case <-p.entered: + case <-time.After(5 * time.Second): + t.Fatal("the conversion never reached the decoder") + } +} + +// releaseAll lets a held conversion finish; safe to call more than once. +func (p *orientationDecodeProbe) releaseAll() { p.once.Do(func() { close(p.release) }) } + +// runOrientationConversion starts one Predict in the background and returns a +// join func. The cleanup is a safety net: a conversion left inside the decoder +// by a failing assertion is always released and joined before the seam is +// restored, so it cannot leak its slot into the next test. +func runOrientationConversion( + ctx context.Context, t *testing.T, model VLM, data []byte, probe *orientationDecodeProbe, +) func() { + t.Helper() + done := make(chan struct{}) + go func() { + defer close(done) + _, _ = model.Predict(ctx, [][]byte{data}, "prompt") + }() + t.Cleanup(func() { + probe.releaseAll() + <-done + }) + return func() { + probe.releaseAll() + <-done + } +} + +// fillOrientationSlot takes the whole budget from the test itself, failing fast +// instead of blocking when an earlier test leaked a slot. +func fillOrientationSlot(t *testing.T) { + t.Helper() + select { + case orientationSlots <- struct{}{}: + default: + t.Fatal("a conversion slot was already held before this test started") + } + t.Cleanup(func() { <-orientationSlots }) +} + +func requireOrientationSlotFree(t *testing.T) { + t.Helper() + require.Zero(t, len(orientationSlots), "the conversion slot must be released") +} + +// A slot belongs to the process, not to the instance that took it: two VLM +// instances built separately (as every model and every configuration change +// does) must share one budget. While the first sits in the decoder, the second +// must forward its bytes instead of starting a conversion of its own. +func TestOrientationSlotIsSharedByEveryVLMInstance(t *testing.T) { + // Pin the switch on: these cases assert what the pass does, so they must + // not depend on whatever the ambient environment exports. + t.Setenv("VLM_IMAGE_ORIENTATION", "") + stored := markedJPEG(t, 6, binary.BigEndian) + expected := append([]byte(nil), stored...) + + firstInner, secondInner := &recordingVLM{}, &recordingVLM{} + first, err := wrapVLMImageOrientation(firstInner, nil) + require.NoError(t, err) + second, err := wrapVLMImageOrientation(secondInner, nil) + require.NoError(t, err) + + probe := holdOrientationDecode(t) + wait := runOrientationConversion(context.Background(), t, first, stored, probe) + probe.waitInsideDecoder(t) + + _, err = second.Predict(context.Background(), [][]byte{stored}, "prompt") + require.NoError(t, err) + + require.Len(t, secondInner.received, 1) + assert.Equal(t, expected, secondInner.received[0], "a full budget must forward the bytes unchanged") + assert.Same(t, &stored[0], &secondInner.received[0][0], "the payload must be forwarded by reference") + assert.Equal(t, int64(1), probe.calls.Load(), "the refused call must not reach the decoder") + + wait() + width, height := decodeSize(t, firstInner.received[0]) + assert.Equal(t, 8, width, "the instance that owns the slot still gets upright pixels") + assert.Equal(t, 24, height) +} + +// The budget is not the background governor's business: an interactive chat +// upload and a background enrichment draw from the same slot, in both +// directions, whichever one of them got there first. +func TestOrientationSlotIsSharedByInteractiveAndBackgroundCalls(t *testing.T) { + // Pin the switch on: these cases assert what the pass does, so they must + // not depend on whatever the ambient environment exports. + t.Setenv("VLM_IMAGE_ORIENTATION", "") + for _, test := range []struct { + name string + holderCtx context.Context + callerCtx context.Context + }{ + { + name: "background conversion refuses an interactive call", + holderCtx: types.WithBackgroundTask(context.Background()), + callerCtx: context.Background(), + }, + { + name: "interactive conversion refuses a background call", + holderCtx: context.Background(), + callerCtx: types.WithBackgroundTask(context.Background()), + }, + } { + t.Run(test.name, func(t *testing.T) { + stored := markedJPEG(t, 6, binary.BigEndian) + expected := append([]byte(nil), stored...) + + holderInner, callerInner := &recordingVLM{}, &recordingVLM{} + holder, err := wrapVLMImageOrientation(holderInner, nil) + require.NoError(t, err) + caller, err := wrapVLMImageOrientation(callerInner, nil) + require.NoError(t, err) + + probe := holdOrientationDecode(t) + wait := runOrientationConversion(test.holderCtx, t, holder, stored, probe) + probe.waitInsideDecoder(t) + + _, err = caller.Predict(test.callerCtx, [][]byte{stored}, "prompt") + require.NoError(t, err) + + require.Len(t, callerInner.received, 1) + assert.Equal(t, expected, callerInner.received[0], "the caller must get its bytes back unchanged") + assert.Same(t, &stored[0], &callerInner.received[0][0]) + assert.Equal(t, int64(1), probe.calls.Load(), "only the holder may decode") + + wait() + }) + } +} + +// With the budget full the payload reaches the model byte for byte, tag +// included, and the decoder is never entered — the same shape as the +// over-budget pixel branch. +func TestOrientationSlotExhaustionForwardsThePayloadUntouched(t *testing.T) { + // Pin the switch on: these cases assert what the pass does, so they must + // not depend on whatever the ambient environment exports. + t.Setenv("VLM_IMAGE_ORIENTATION", "") + stored := markedJPEG(t, 6, binary.BigEndian) + expected := append([]byte(nil), stored...) + + fillOrientationSlot(t) // someone else in the process owns the only slot + probe := countOrientationDecodes(t) + + out := orientImageBytes(context.Background(), stored) + + assert.Equal(t, expected, out, "the bytes must come back unchanged") + assert.Same(t, &stored[0], &out[0], "the payload must come back by reference") + assert.Equal(t, 6, jpegEXIFOrientation(out), "the rotation tag must survive the passthrough") + assert.Zero(t, probe.calls.Load(), "an exhausted budget must not decode anything") +} + +// Every return path must give the slot back. A leak would not fail loudly: it +// would silently turn normalisation off for every later call in the process, so +// each case proves that the next conversion still gets its slot. +func TestOrientationSlotIsReleasedOnEveryReturnPath(t *testing.T) { + // Pin the switch on: these cases assert what the pass does, so they must + // not depend on whatever the ambient environment exports. + t.Setenv("VLM_IMAGE_ORIENTATION", "") + stored := markedJPEG(t, 6, binary.BigEndian) + + t.Run("conversion succeeds", func(t *testing.T) { + probe := countOrientationDecodes(t) + + oriented := orientImageBytes(context.Background(), stored) + width, height := decodeSize(t, oriented) + require.Equal(t, []int{8, 24}, []int{width, height}) + require.Equal(t, int64(1), probe.calls.Load()) + requireOrientationSlotFree(t) + + again := orientImageBytes(context.Background(), stored) + againWidth, againHeight := decodeSize(t, again) + assert.Equal(t, []int{8, 24}, []int{againWidth, againHeight}, + "a finished conversion must leave its slot available") + }) + + t.Run("decode fails", func(t *testing.T) { + // A real payload cut just past its scan header: the header probe is + // happy, the full decode is not. + broken := truncatedAfterHeader(t, stored) + require.Equal(t, 6, jpegEXIFOrientation(broken)) + _, _, configErr := image.DecodeConfig(bytes.NewReader(broken)) + require.NoError(t, configErr, "the fixture must get past the pixel-budget probe") + _, _, decodeErr := image.Decode(bytes.NewReader(broken)) + require.Error(t, decodeErr, "the fixture must fail the full decode") + + out := orientImageBytes(context.Background(), broken) + assert.Same(t, &broken[0], &out[0], "a failed decode must forward the stored bytes") + requireOrientationSlotFree(t) + + next := orientImageBytes(context.Background(), stored) + nextWidth, nextHeight := decodeSize(t, next) + assert.Equal(t, []int{8, 24}, []int{nextWidth, nextHeight}, + "a failed conversion must leave its slot available") + }) +} + +// truncatedAfterHeader cuts a JPEG just past its start-of-scan marker: enough +// for image.DecodeConfig, not enough for image.Decode. +func truncatedAfterHeader(t *testing.T, data []byte) []byte { + t.Helper() + for i := 2; i+1 < len(data); i++ { + if data[i] == 0xFF && data[i+1] == jpegMarkerSOS { + return data[:i+4] + } + } + t.Fatal("fixture has no start-of-scan marker") + return nil +} diff --git a/internal/models/vlm/image_orientation_switch_test.go b/internal/models/vlm/image_orientation_switch_test.go new file mode 100644 index 00000000000..36921684076 --- /dev/null +++ b/internal/models/vlm/image_orientation_switch_test.go @@ -0,0 +1,49 @@ +package vlm + +import ( + "bytes" + "context" + "encoding/binary" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// TestOrientationCanBeTurnedOff pins the escape hatch: a deployment whose +// sources carry untrustworthy EXIF tags must be able to get the stored bytes +// forwarded untouched, because honouring a wrong tag turns correct pixels +// sideways. +func TestOrientationCanBeTurnedOff(t *testing.T) { + stored := markedJPEG(t, 6, binary.BigEndian) + ctx := context.Background() + + t.Run("default keeps normalising", func(t *testing.T) { + // Pin it to the default explicitly: the switch reads the ambient + // environment, so a developer (or a CI job) that exports + // VLM_IMAGE_ORIENTATION=off would otherwise flip this case over. + t.Setenv("VLM_IMAGE_ORIENTATION", "") + out := orientImageBytes(ctx, stored) + require.NotEqual(t, stored, out, "the tag must be honoured by default") + assert.Zero(t, jpegEXIFOrientation(out), "the re-encoded page carries no tag") + }) + + for _, value := range []string{"0", "false", "no", "off", " OFF "} { + t.Run("disabled by "+value, func(t *testing.T) { + t.Setenv("VLM_IMAGE_ORIENTATION", value) + out := orientImageBytes(ctx, stored) + assert.True(t, bytes.Equal(stored, out), + "a disabled pass must forward the stored bytes byte for byte") + assert.Equal(t, 6, jpegEXIFOrientation(out), + "the stored tag must survive untouched") + }) + } + + for _, value := range []string{"1", "true", "on", "yes", ""} { + t.Run("still on for "+value, func(t *testing.T) { + t.Setenv("VLM_IMAGE_ORIENTATION", value) + out := orientImageBytes(ctx, stored) + assert.NotEqual(t, stored, out, "anything but the off spellings keeps the pass on") + }) + } +} diff --git a/internal/models/vlm/image_orientation_test.go b/internal/models/vlm/image_orientation_test.go new file mode 100644 index 00000000000..7daf0e43bea --- /dev/null +++ b/internal/models/vlm/image_orientation_test.go @@ -0,0 +1,321 @@ +package vlm + +import ( + "bytes" + "context" + "encoding/binary" + "image" + "image/color" + "image/jpeg" + "image/png" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// markedJPEG builds a JPEG whose only bright spot is a block in the top-left +// corner, then writes an EXIF Orientation into an APP1 segment in front of it — +// exactly how a camera or scanner stores a page it captured sideways. +func markedJPEG(t *testing.T, orientation int, order binary.ByteOrder) []byte { + t.Helper() + const width, height, block = 24, 8, 4 + src := image.NewRGBA(image.Rect(0, 0, width, height)) + for y := range height { + for x := range width { + src.Set(x, y, color.RGBA{A: 255}) + } + } + for y := range block { + for x := range block { + src.Set(x, y, color.RGBA{R: 255, G: 255, B: 255, A: 255}) + } + } + var buf bytes.Buffer + require.NoError(t, jpeg.Encode(&buf, src, &jpeg.Options{Quality: 100})) + raw := buf.Bytes() + app1 := exifAPP1(orientation, order) + out := make([]byte, 0, len(raw)+len(app1)) + out = append(out, raw[:2]...) // SOI + out = append(out, app1...) + out = append(out, raw[2:]...) + return out +} + +// exifAPP1 encodes a minimal APP1 segment ("Exif\0\0" + TIFF header + one IFD0 +// entry) carrying only the Orientation tag. +func exifAPP1(orientation int, order binary.ByteOrder) []byte { + tiff := make([]byte, 26) + if order == binary.LittleEndian { + copy(tiff, "II") + } else { + copy(tiff, "MM") + } + order.PutUint16(tiff[2:4], 42) // TIFF magic + order.PutUint32(tiff[4:8], 8) // IFD0 offset + order.PutUint16(tiff[8:10], 1) // one entry + order.PutUint16(tiff[10:12], exifOrientationTag) // 0x0112 + order.PutUint16(tiff[12:14], 3) // SHORT + order.PutUint32(tiff[14:18], 1) // count + order.PutUint16(tiff[18:20], uint16(orientation)) // value + payload := append([]byte("Exif\x00\x00"), tiff...) // next IFD offset stays 0 + segmentLength := len(payload) + 2 + segment := []byte{0xFF, jpegMarkerAPP1, byte(segmentLength >> 8), byte(segmentLength)} + return append(segment, payload...) +} + +func decodeSize(t *testing.T, data []byte) (int, int) { + t.Helper() + img, _, err := image.Decode(bytes.NewReader(data)) + require.NoError(t, err) + cfg := img.Bounds() + return cfg.Dx(), cfg.Dy() +} + +// brightestCorner reports which 4x4 corner of the decoded image is bright. The +// fixture paints exactly one corner, so this is how the tests tell a 90 degree +// clockwise turn from a counter-clockwise one. +func brightestCorner(t *testing.T, data []byte) string { + t.Helper() + img, _, err := image.Decode(bytes.NewReader(data)) + require.NoError(t, err) + bounds := img.Bounds() + const block = 4 + corners := []struct { + name string + x0, y0 int + }{ + {"top-left", bounds.Min.X, bounds.Min.Y}, + {"top-right", bounds.Max.X - block, bounds.Min.Y}, + {"bottom-left", bounds.Min.X, bounds.Max.Y - block}, + {"bottom-right", bounds.Max.X - block, bounds.Max.Y - block}, + } + best, bestMean := "", -1.0 + for _, corner := range corners { + sum := 0.0 + for y := range block { + for x := range block { + r, g, b, _ := img.At(corner.x0+x, corner.y0+y).RGBA() + sum += float64(r+g+b) / 3 + } + } + if mean := sum / (block * block); mean > bestMean { + best, bestMean = corner.name, mean + } + } + return best +} + +func TestOrientImageBytesTurnsPixelsPerEXIFTag(t *testing.T) { + // Pin the switch on: these cases assert what the pass does, so they must + // not depend on whatever the ambient environment exports. + t.Setenv("VLM_IMAGE_ORIENTATION", "") + for _, test := range []struct { + orientation int + width int + height int + brightest string + }{ + {orientation: 2, width: 24, height: 8, brightest: "top-right"}, + {orientation: 3, width: 24, height: 8, brightest: "bottom-right"}, + {orientation: 4, width: 24, height: 8, brightest: "bottom-left"}, + {orientation: 5, width: 8, height: 24, brightest: "top-left"}, + {orientation: 6, width: 8, height: 24, brightest: "top-right"}, + {orientation: 7, width: 8, height: 24, brightest: "bottom-right"}, + {orientation: 8, width: 8, height: 24, brightest: "bottom-left"}, + } { + t.Run(string(rune('0'+test.orientation)), func(t *testing.T) { + stored := markedJPEG(t, test.orientation, binary.BigEndian) + require.Equal(t, test.orientation, jpegEXIFOrientation(stored)) + + oriented := orientImageBytes(context.Background(), stored) + + width, height := decodeSize(t, oriented) + assert.Equal(t, test.width, width, "width") + assert.Equal(t, test.height, height, "height") + assert.Equal(t, test.brightest, brightestCorner(t, oriented), + "orientation %d must move the captured top-left corner to %s", test.orientation, test.brightest) + // The pixels are upright now, so the tag must not survive: a viewer + // that honoured it would turn the page a second time. + assert.Zero(t, jpegEXIFOrientation(oriented), "the orientation tag must be gone") + }) + } +} + +func TestOrientImageBytesReadsBothTIFFByteOrders(t *testing.T) { + // Pin the switch on: these cases assert what the pass does, so they must + // not depend on whatever the ambient environment exports. + t.Setenv("VLM_IMAGE_ORIENTATION", "") + for name, order := range map[string]binary.ByteOrder{ + "big-endian": binary.BigEndian, + "little-endian": binary.LittleEndian, + } { + t.Run(name, func(t *testing.T) { + stored := markedJPEG(t, 6, order) + require.Equal(t, 6, jpegEXIFOrientation(stored)) + + width, height := decodeSize(t, orientImageBytes(context.Background(), stored)) + assert.Equal(t, 8, width) + assert.Equal(t, 24, height) + }) + } +} + +func TestOrientImageBytesLeavesUntouchedPayloadsAlone(t *testing.T) { + // Pin the switch on: these cases assert what the pass does, so they must + // not depend on whatever the ambient environment exports. + t.Setenv("VLM_IMAGE_ORIENTATION", "") + var pngBuf bytes.Buffer + require.NoError(t, png.Encode(&pngBuf, image.NewRGBA(image.Rect(0, 0, 8, 4)))) + + truncated := markedJPEG(t, 6, binary.BigEndian)[:40] + corrupt := markedJPEG(t, 6, binary.BigEndian) + copy(corrupt[24+6:24+10], []byte{0xDE, 0xAD, 0xBE, 0xEF}) // break the TIFF header + + for name, payload := range map[string][]byte{ + "orientation 1": markedJPEG(t, 1, binary.BigEndian), + "no exif": markedJPEGWithoutEXIF(t), + "png": pngBuf.Bytes(), + "truncated": truncated, + "corrupt tiff": corrupt, + "empty": {}, + "not an image": []byte("this is not an image at all"), + "orientation 0": markedJPEG(t, 0, binary.BigEndian), + "orientation 9": markedJPEG(t, 9, binary.BigEndian), + } { + t.Run(name, func(t *testing.T) { + out := orientImageBytes(context.Background(), payload) + require.Len(t, out, len(payload)) + if len(payload) > 0 { + assert.Same(t, &payload[0], &out[0], "the payload must be passed through by reference") + } + }) + } +} + +func TestOrientImageBytesSkipsCanvasesOverThePixelBudget(t *testing.T) { + // Pin the switch on: these cases assert what the pass does, so they must + // not depend on whatever the ambient environment exports. + t.Setenv("VLM_IMAGE_ORIENTATION", "") + previous := maxOrientationPixels + maxOrientationPixels = 16 + t.Cleanup(func() { maxOrientationPixels = previous }) + + stored := markedJPEG(t, 6, binary.BigEndian) + out := orientImageBytes(context.Background(), stored) + + require.Len(t, out, len(stored)) + assert.Same(t, &stored[0], &out[0], + "an oversized canvas keeps its stored orientation instead of exhausting the worker") +} + +func TestOrientationVLMPredictsWithUprightPixels(t *testing.T) { + // Pin the switch on: these cases assert what the pass does, so they must + // not depend on whatever the ambient environment exports. + t.Setenv("VLM_IMAGE_ORIENTATION", "") + inner := &recordingVLM{} + model, err := wrapVLMImageOrientation(inner, nil) + require.NoError(t, err) + + stored := markedJPEG(t, 6, binary.BigEndian) + original := append([]byte(nil), stored...) + _, err = model.Predict(context.Background(), [][]byte{stored}, "prompt") + require.NoError(t, err) + + require.Len(t, inner.received, 1) + width, height := decodeSize(t, inner.received[0]) + assert.Equal(t, 8, width, "the model must receive the rotated page") + assert.Equal(t, 24, height) + assert.Equal(t, "top-right", brightestCorner(t, inner.received[0])) + assert.Equal(t, original, stored, "the caller's bytes must not be rewritten") +} + +func TestOrientationVLMPassesPayloadsThroughUntouched(t *testing.T) { + // Pin the switch on: these cases assert what the pass does, so they must + // not depend on whatever the ambient environment exports. + t.Setenv("VLM_IMAGE_ORIENTATION", "") + inner := &recordingVLM{} + model, err := wrapVLMImageOrientation(inner, nil) + require.NoError(t, err) + + stored := markedJPEGWithoutEXIF(t) + _, err = model.Predict(context.Background(), [][]byte{stored}, "prompt") + require.NoError(t, err) + require.Len(t, inner.received, 1) + assert.Same(t, &stored[0], &inner.received[0][0]) + + _, err = model.Predict(context.Background(), nil, "prompt") + require.NoError(t, err) + assert.Nil(t, inner.received) +} + +func markedJPEGWithoutEXIF(t *testing.T) []byte { + t.Helper() + var buf bytes.Buffer + require.NoError(t, jpeg.Encode(&buf, image.NewRGBA(image.Rect(0, 0, 8, 4)), &jpeg.Options{Quality: 100})) + return buf.Bytes() +} + +// markedJPEGAfterJFIFApp0 stores the EXIF block behind a JFIF APP0 — the layout +// every camera and scanner writes (Go's own encoder emits no APP0), and the one +// that catches a parser which assumes EXIF is the very first segment. +func markedJPEGAfterJFIFApp0(t *testing.T, orientation int, order binary.ByteOrder) []byte { + t.Helper() + raw := markedJPEG(t, orientation, order) + app1 := exifAPP1(orientation, order) + // Drop the APP1 the fixture above injected, then re-insert it behind APP0. + withoutAPP1 := append(append([]byte{}, raw[:2]...), raw[2+len(app1):]...) + app0 := []byte{ + 0xFF, 0xE0, 0x00, 0x10, // APP0, 16 bytes including the length field + 'J', 'F', 'I', 'F', 0x00, + 0x01, 0x01, 0x00, 0x00, 0x01, 0x00, 0x01, 0x00, 0x00, + } + out := make([]byte, 0, len(withoutAPP1)+len(app0)+len(app1)) + out = append(out, withoutAPP1[:2]...) + out = append(out, app0...) + out = append(out, app1...) + out = append(out, withoutAPP1[2:]...) + return out +} + +func TestOrientImageBytesFindsEXIFBehindJFIFApp0(t *testing.T) { + // Pin the switch on: these cases assert what the pass does, so they must + // not depend on whatever the ambient environment exports. + t.Setenv("VLM_IMAGE_ORIENTATION", "") + stored := markedJPEGAfterJFIFApp0(t, 6, binary.BigEndian) + require.Equal(t, byte(0xE0), stored[3], "APP0 must come first in this fixture") + require.Equal(t, byte(0xE1), stored[21], "EXIF must sit behind APP0") + require.Equal(t, 6, jpegEXIFOrientation(stored), "the tag must be found after APP0") + + width, height := decodeSize(t, orientImageBytes(context.Background(), stored)) + assert.Equal(t, 8, width) + assert.Equal(t, 24, height) +} + +// Normalising twice must be a passthrough the second time: the first pass +// removes the tag with the pixels, so a second decode would only burn CPU. +func TestOrientImageBytesIsIdempotent(t *testing.T) { + // Pin the switch on: these cases assert what the pass does, so they must + // not depend on whatever the ambient environment exports. + t.Setenv("VLM_IMAGE_ORIENTATION", "") + stored := markedJPEG(t, 6, binary.BigEndian) + + once := orientImageBytes(context.Background(), stored) + twice := orientImageBytes(context.Background(), once) + + require.Len(t, twice, len(once)) + assert.Same(t, &once[0], &twice[0], "the second pass must return the payload by reference") + assert.Zero(t, jpegEXIFOrientation(twice)) +} + +type recordingVLM struct { + received [][]byte +} + +func (r *recordingVLM) Predict(_ context.Context, imgBytes [][]byte, _ string) (string, error) { + r.received = imgBytes + return "ok", nil +} + +func (r *recordingVLM) GetModelName() string { return "recording" } +func (r *recordingVLM) GetModelID() string { return "recording" } diff --git a/internal/models/vlm/vlm.go b/internal/models/vlm/vlm.go index 42827aa4810..adfac4214b8 100644 --- a/internal/models/vlm/vlm.go +++ b/internal/models/vlm/vlm.go @@ -99,6 +99,15 @@ func NewVLM(config *Config, ollamaService *ollama.OllamaService) (VLM, error) { v = &debugVLM{inner: v} } v, err = wrapVLMLangfuse(v, nil) + // Outside the trace layers (debug / Langfuse record the payload they are + // handed, so normalising further in would leave the trace showing bytes the + // model never received), and inside the per-model gate below, which only + // covers background calls: the decode and re-encode of a 600 DPI page is + // bounded by the process-wide rotation budget instead (see the + // acquireOrientationSlot gate). Models read the pixel matrix and ignore the + // EXIF orientation tag that browsers honour, so the pixels are turned + // upright here. + v, _ = wrapVLMImageOrientation(v, nil) // Outermost: hold the per-model concurrency slot only around the real // provider round-trip, so the wait is excluded from debug/langfuse timing. return wrapVLMConcurrency(v, config.MaxConcurrency, err) diff --git a/website-docs/01-getting-started/04-configuration.md b/website-docs/01-getting-started/04-configuration.md index 6780be473bd..5451dc33de6 100644 --- a/website-docs/01-getting-started/04-configuration.md +++ b/website-docs/01-getting-started/04-configuration.md @@ -212,6 +212,7 @@ AWS S3 的 `S3_ACCESS_KEY` / `S3_SECRET_KEY` 可以**同时留空**,此时走 | `OLLAMA_OPTIONAL` | true | Ollama 不可用时仅告警不阻断启动 | | `BATCH_EMBED_SIZE` | 空 | 批量 embedding 大小 | | `VLM_HTTP_TIMEOUT_SECONDS` | 180 | VLM 单次请求超时 | +| `VLM_IMAGE_ORIENTATION` | 开启 | 按 EXIF 方向把图片像素转正后再送模型;设为 `off` 则按存储原样转发(EXIF 标签可能被相机写错,见下)| | `BUILTIN_MODELS_CONFIG` | config/builtin_models.yaml | 内置模型声明文件路径(见下文) | | `MODELS_CONFIG` | config/models.json | 模型厂商目录的部署叠加文件路径(补充厂商、覆盖地址或模型参数),格式见[模型管理](../03-features/06-models.md) | | `WEKNORA_LLM_STREAM_RAW_DUMP` / `_DIR` | 空 | LLM 流原始转储(排障用) |