Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
50 commits
Select commit Hold shift + click to select a range
00dd7d3
fix(assistant): correct data-dictionary drift vs the migration
nedda76 Jul 9, 2026
200404f
fix(assistant): keep hard data-traps under RAG and floor low-relevanc…
nedda76 Jul 9, 2026
5f0bdc7
fix(assistant): block string-building aggregates in the SQL scalar guard
nedda76 Jul 9, 2026
0d15cc8
fix(assistant): harden report emission integrity
nedda76 Jul 9, 2026
7dd5035
refactor(assistant): single-source the data-trap rendering across bot…
nedda76 Jul 9, 2026
025a700
fix(assistant): treat a scoreless RAG match as below the relevance floor
nedda76 Jul 11, 2026
46045b8
fix(assistant): match spelled magnitudes by -илион/-илиард suffix (co…
nedda76 Jul 11, 2026
6b45133
fix(assistant): block string_agg (SQLite ≥3.44 group_concat alias) in…
nedda76 Jul 11, 2026
c4b61a9
fix(assistant): short-circuit validateEmitShape on over-cap arrays
nedda76 Jul 22, 2026
0e2aba9
fix(assistant): stop double-rendering data traps between hardTraps an…
nedda76 Aug 18, 2026
8cf5c83
fix(assistant): version the schema corpus via native Vectorize namesp…
nedda76 Aug 18, 2026
6f7de1f
test(assistant): close the sweep gaps around the corpus-version guard
nedda76 Aug 18, 2026
1c1bec6
docs(assistant): уточни бележката за near-collisions при суфиксите на…
nedda76 Aug 19, 2026
42a00c2
fix(assistant): флагвай и абревиатурите млрд/млн като стемове в prose…
nedda76 Aug 19, 2026
15e34ac
fix(assistant): затвори quoted-identifier bypass-а на функционалния d…
nedda76 Aug 19, 2026
38acf86
fix(assistant): jsonb_group_* влиза в денилиста — JSONB близнаците ми…
nedda76 Sep 2, 2026
fd2a80f
fix(assistant): кавичките на идентификаторите са непрозрачни, а денил…
nedda76 Sep 2, 2026
c6b5ed0
test(assistant): закови fail-closed пътищата на guard-а — незатворен …
nedda76 Sep 2, 2026
fbda389
fix(assistant): лексикалните проверки да не четат съдържанието на стр…
nedda76 Sep 5, 2026
e7fb45b
fix(assistant): премести semantic_search на native Vectorize namespac…
nedda76 Aug 18, 2026
63f2491
fix(assistant): закали entity namespace прехода по бележките от ревюто
nedda76 Aug 18, 2026
1aea6b9
fix(assistant): релевантен флор за semantic_search — симетричен на сх…
nedda76 Aug 19, 2026
943bf1d
fix(assistant): scoreless match отпада при всеки флор + README за pre…
nedda76 Aug 20, 2026
20380a9
fix(assistant): изравни и схема флора на Number.isFinite (безопасност…
nedda76 Sep 1, 2026
ba3a5fa
fix(assistant): типизирай AI/Vectorize биндингите — без 'as unknown a…
nedda76 Aug 18, 2026
ee667cd
fix(assistant): закали типизираните биндинги по бележките от ревюто
nedda76 Aug 18, 2026
45b2d43
fix(assistant): адаптерът отхвърля празен data масив вместо да го чет…
nedda76 Aug 19, 2026
9a2848a
docs(assistant): embed() назовава контракта на адаптера за празен вход
nedda76 Sep 2, 2026
2c9f2c7
feat(assistant): направи RAG fallback-а наблюдаем — статистика на ret…
nedda76 Aug 18, 2026
a45b09f
fix(assistant): закали статистиката на retrieval-а по бележките от ре…
nedda76 Aug 18, 2026
49c8ace
feat(assistant): симетрична статистика и за semantic_search + Number.…
nedda76 Aug 20, 2026
c3915b4
fix(assistant): единен leak-safe лог на грешки във всички catch-ове н…
nedda76 Aug 21, 2026
b20a5ce
fix(assistant): направи errorText тотален и редактиращ — досегашният …
nedda76 Aug 21, 2026
baa95a3
fix(assistant): свивай whitespace-а и на needle-а за редакция, не сам…
nedda76 Aug 23, 2026
11bb845
fix(assistant): stackHead дава само рамки, а редакцията лови и escape…
nedda76 Aug 23, 2026
ffcbfe9
fix(assistant): затвори трите останали ехо-пътя на въпроса към лога +…
nedda76 Aug 23, 2026
7f24425
fix(assistant): павилион не е число, а валидаторът затваря sub/null/link
nedda76 Aug 23, 2026
3aafce7
fix(assistant): речникът описва капаните на анексите; „млн./млрд." фл…
nedda76 Aug 23, 2026
d524d1d
fix(assistant): обезвреди async stats sink, закови returnMetadata, оп…
nedda76 Aug 23, 2026
7c8af91
style(assistant): prettier върху файловете от ревюто
nedda76 Aug 23, 2026
b96905a
fix(assistant): „трлн" в абревиатурния клон + дедуп и на примитивни s…
nedda76 Sep 1, 2026
b3e9023
fix(assistant): prose gate-ът е fail-closed за млн/млрд/трлн — затвор…
nedda76 Sep 2, 2026
7e367b1
test(assistant): „pre-cap" тестът на errorText наистина да минава пре…
nedda76 Sep 2, 2026
125b8a2
test(assistant): покрий RetryError без API грешка, късия needle, emoj…
nedda76 Sep 2, 2026
b053bd1
fix(assistant): закови списъка полета на изричното изграждане на коло…
nedda76 Sep 5, 2026
49faf6c
test(assistant): закови речника на данните срещу реалните миграции (д…
nedda76 Aug 23, 2026
dd0f2b2
fix(assistant): редактирай ВСЕКИ потребителски ход, не само последния…
nedda76 Sep 1, 2026
dacc381
test(assistant): дрифт-пазачът да не пропуска не-ASCII име на колона
nedda76 Sep 5, 2026
b8fd514
feat(assistant): самопровизиониране на schema-v2 корпуса + GET /assis…
nedda76 Sep 5, 2026
f55eb95
ci(deploy): проверка на корпуса на асистента след deploy (#346)
nedda76 Sep 5, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
44 changes: 44 additions & 0 deletions .github/workflows/deploy.yml
Original file line number Diff line number Diff line change
Expand Up @@ -489,3 +489,47 @@ jobs:
[ -n "$value" ] || { echo "$name is empty; leaving the Worker secret unchanged."; continue; }
printf '%s' "$value" | pnpm --filter @sigma/etl exec wrangler secret put "$name" --config wrangler.deploy.toml
done

# #346: prove the assistant's schema corpus is indexed in the environment just deployed. LAST in the
# job on purpose: a red check must not leave the explorer and the ETL Workflow at different versions
# or skip the LOG_IP_KEY initialisation above — everything else has already shipped by now. The Worker
# provisions the corpus itself on first use (ensureSchemaCorpus, #328); GET /assistant/health is both
# the trigger and the proof — 200 only when every vector THIS build expects is readable in `schema-v2`
# with this build's text. Vectorize applies writes asynchronously, so the first call after a fresh
# deploy legitimately answers 503 with `upserted` > 0; retry for ~2 minutes before failing the deploy.
# Both environments sit behind Cloudflare Access, so the probe needs an Access SERVICE TOKEN
# (Environment secrets CF_ACCESS_CLIENT_ID / CF_ACCESS_CLIENT_SECRET) and the environment's public base
# URL (Environment variable SIGMA_WEB_URL). Until an environment defines SIGMA_WEB_URL the step is
# skipped with a notice — the deploy must never depend on configuration the environment was not given
# (docs/deploy.md §5). Only counters and an HTTP status are printed; the health body carries no text.
- name: Verify assistant schema corpus (#346)
if: steps.guard.outputs.ok == 'true'
env:
SIGMA_WEB_URL: ${{ vars.SIGMA_WEB_URL }}
CF_ACCESS_CLIENT_ID: ${{ secrets.CF_ACCESS_CLIENT_ID }}
CF_ACCESS_CLIENT_SECRET: ${{ secrets.CF_ACCESS_CLIENT_SECRET }}
run: |
if [ -z "$SIGMA_WEB_URL" ]; then
echo "::notice::SIGMA_WEB_URL is not set for this environment; skipping the assistant corpus check (#346)."
exit 0
fi
url="${SIGMA_WEB_URL%/}/assistant/health"
auth=()
if [ -n "$CF_ACCESS_CLIENT_ID" ] && [ -n "$CF_ACCESS_CLIENT_SECRET" ]; then
auth=(-H "CF-Access-Client-Id: $CF_ACCESS_CLIENT_ID" -H "CF-Access-Client-Secret: $CF_ACCESS_CLIENT_SECRET")
fi
if [ -n "$CF_ACCESS_CLIENT_ID$CF_ACCESS_CLIENT_SECRET" ] && [ "${#auth[@]}" = 0 ]; then
echo "::warning::Only one of CF_ACCESS_CLIENT_ID / CF_ACCESS_CLIENT_SECRET is set; probing WITHOUT an Access service token."
fi
for attempt in 1 2 3 4 5 6 7 8; do
: > /tmp/health.json
# curl's -w prints 000 itself on a transport failure; `|| true` only keeps `set -e` from
# aborting the loop. The body is counters-only JSON on success; anything else (an Access
# login page, a 5xx) is cut to one short line so the log never carries a page.
code="$(curl -sS "${auth[@]}" -o /tmp/health.json -w '%{http_code}' "$url" || true)"
echo "attempt $attempt: HTTP ${code:-000} $(head -c 300 /tmp/health.json | tr '\n' ' ')"
[ "$code" = "200" ] && exit 0
[ "$attempt" = 8 ] || sleep 15
done
echo "::error::assistant schema corpus is not fully indexed at $url after ~2 minutes (#346) — see the attempts above."
exit 1
91 changes: 66 additions & 25 deletions apps/web/app/lib/assistant/README.md

Large diffs are not rendered by default.

93 changes: 93 additions & 0 deletions apps/web/app/lib/assistant/agent.stream-error.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,93 @@
import { afterEach, describe, expect, it, vi } from 'vitest';
import { APICallError } from 'ai';
import { MockLanguageModelV3 } from 'ai/test';
import { runAssistant } from './agent';
import type { ToolContext } from './tools';

// The ONLY test that drives runAssistant through the real streamText loop. It exists for one
// invariant the unit tests cannot see: the AI SDK's own DEFAULT `onError` is `console.error(error)`
// — the RAW APICallError, whose own properties carry `requestBodyValues` (system prompt + the user's
// messages) and `responseBody`. If agent.ts stops overriding it on streamText (not only on the UI
// stream), the prompt lands in the tail log on every provider failure, and nothing else notices.
const h = vi.hoisted(() => ({ model: undefined as unknown }));
vi.mock('@ai-sdk/openai', () => ({
createOpenAI: () => ({ chat: () => h.model }),
}));

const QUESTION = 'колко плати община Пловдив на фирма Х през 2024';

function failingModel(statusCode: number, isRetryable: boolean) {
return new MockLanguageModelV3({
doStream: async () => {
throw new APICallError({
// A provider body that quotes the prompt back — the echo sits in `.message`.
message: `upstream rejected: ${QUESTION}`,
url: 'https://api.bggpt.ai/v1/chat/completions',
requestBodyValues: { messages: [{ role: 'user', content: QUESTION }] },
statusCode,
responseBody: `{"error":{"message":"${QUESTION}"}}`,
isRetryable,
});
},
});
}

function ctx(): ToolContext {
return {
db: {} as never,
results: [],
rowsRead: 0,
rowsReadBudget: 1000,
userQuestion: QUESTION,
};
}

async function runToEnd(): Promise<string> {
const res = await runAssistant({
env: { BGGPT_API_KEY: 'k', MAX_STEPS: '1' },
ctx: ctx(),
messages: [{ id: 'u1', role: 'user', parts: [{ type: 'text', text: QUESTION }] }],
});
return res.text(); // draining the body drives the stream (and its error hooks) to completion
}

describe('runAssistant stream error logging', () => {
afterEach(() => vi.restoreAllMocks());

it('logs ONE redacted, tagged line and NEVER the raw error object (non-retryable 401)', async () => {
h.model = failingModel(401, false);
const error = vi.spyOn(console, 'error').mockImplementation(() => {});
const body = await runToEnd();

// What the client sees: our line, not the provider's.
expect(body).toContain('Асистентът временно не е достъпен');
expect(body).not.toContain('Пловдив');

// What the tail log sees: exactly one line, a string, redacted, with identifier-only context.
expect(error).toHaveBeenCalledTimes(1);
const [first] = error.mock.calls[0] ?? [];
expect(typeof first).toBe('string'); // the SDK default would pass the Error OBJECT here
expect(first).toMatch(/^\[assistant\] stream error: /);
expect(first).toContain('«редактирано»');
expect(first).toContain('status=401');
expect(first).toContain('retryable=false');
for (const call of error.mock.calls) {
for (const arg of call) {
expect(typeof arg).toBe('string');
expect(String(arg)).not.toContain('Пловдив');
expect(String(arg)).not.toContain('requestBodyValues');
}
}
});

it('keeps the status visible through the RetryError wrapper (retryable 429, maxRetries 1)', async () => {
h.model = failingModel(429, true);
const error = vi.spyOn(console, 'error').mockImplementation(() => {});
await runToEnd();
expect(error).toHaveBeenCalledTimes(1);
const line = String(error.mock.calls[0]?.[0]);
expect(line).toContain('RetryError maxRetriesExceeded');
expect(line).toContain('status=429');
expect(line).not.toContain('Пловдив');
});
});
136 changes: 133 additions & 3 deletions apps/web/app/lib/assistant/agent.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,9 @@ import { fakeD1 } from '@sigma/test-support';

// agent.ts is thin Vercel-AI-SDK wiring. Mock the SDK and provider so the tests can assert the wiring
// (model/base-URL resolution, tool-set assembly, stream Response + onError message) without a live
// BgGPT call. resolveMaxSteps is pure and needs no mocks.
// BgGPT call. resolveMaxSteps is pure and needs no mocks. The `ai` mock SPREADS the real module:
// agent.ts also imports APICallError/RetryError from it for the stream-error logger, and a mock that
// replaced the module wholesale would turn those into `undefined` and crash every onError path.
const { streamTextMock, createOpenAIMock, chatMock } = vi.hoisted(() => {
const chatMock = vi.fn((model: string) => ({ model }));
return {
Expand All @@ -13,15 +15,17 @@ const { streamTextMock, createOpenAIMock, chatMock } = vi.hoisted(() => {
};
});
vi.mock('@ai-sdk/openai', () => ({ createOpenAI: createOpenAIMock }));
vi.mock('ai', () => ({
vi.mock('ai', async (importOriginal) => ({
...(await importOriginal<typeof import('ai')>()),
convertToModelMessages: vi.fn(async (m: unknown) => m),
jsonSchema: vi.fn((s: unknown) => s),
stepCountIs: vi.fn((n: number) => ({ stopAt: n })),
streamText: (o: unknown) => streamTextMock(o),
tool: (def: unknown) => def,
}));

import { resolveMaxSteps, runAssistant } from './agent';
import { RetryError } from 'ai';
import { makeStreamErrorLogger, resolveMaxSteps, runAssistant, userTexts } from './agent';
import { ASSISTANT_TOOLS } from './tools';

describe('resolveMaxSteps', () => {
Expand Down Expand Up @@ -108,6 +112,15 @@ describe('runAssistant (SDK wiring)', () => {
const r = await tools.emit_report.execute({ not: 'a valid report' });
expect(r.ok).toBe(false);
expect(Array.isArray(r.errors)).toBe(true);

// …and a valid report takes the ok branch, returning the bound (server-owned) report.
const ok = await tools.emit_report.execute({
title: 'Справка',
question: 'въпрос',
blocks: [{ type: 'text', md: 'Няма данни.' }],
});
expect(ok.ok).toBe(true);
expect(ok.report.title).toBe('Справка');
});
});

Expand Down Expand Up @@ -142,3 +155,120 @@ describe('runAssistant — the emit_report tool', () => {
});
});
});

describe('makeStreamErrorLogger', () => {
const capture = () => {
const lines: string[] = [];
const spy = vi.spyOn(console, 'error').mockImplementation((l: unknown) => {
lines.push(String(l));
});
return { lines, restore: () => spy.mockRestore() };
};

it('logs an object error once even though both stream hooks report it', () => {
const { lines, restore } = capture();
const log = makeStreamErrorLogger([]);
const err = new Error('провайдърът падна');
log(err);
log(err); // the second hook sees the SAME object
restore();
expect(lines).toHaveLength(1);
expect(lines[0]).toContain('провайдърът падна');
});

it('logs a PRIMITIVE throw once too — it has no identity for a WeakSet to key on', () => {
const { lines, restore } = capture();
const log = makeStreamErrorLogger([]);
log('низова грешка');
log('низова грешка');
restore();
expect(lines).toHaveLength(1);
});

it('still logs two DISTINCT object errors that render the same text', () => {
const { lines, restore } = capture();
const log = makeStreamErrorLogger([]);
log(new Error('една и съща фраза'));
log(new Error('една и съща фраза'));
restore();
expect(lines).toHaveLength(2);
});

it('tags a RetryError that wraps a NON-API error with its reason only (no status to show)', () => {
const { lines, restore } = capture();
const log = makeStreamErrorLogger([]);
log(
new RetryError({
message: 'опитите свършиха',
reason: 'errorNotRetryable',
errors: [new Error('мрежата падна')],
}),
);
restore();
expect(lines).toHaveLength(1);
expect(lines[0]).toContain('[RetryError errorNotRetryable]');
expect(lines[0]).not.toContain('APICallError');
});

it('redacts the question before it reaches the log line', () => {
const { lines, restore } = capture();
const question = 'колко плати община Пловдив на фирма Х';
makeStreamErrorLogger([question])(new Error(`400: ${question}`));
restore();
expect(lines[0]).not.toContain('Пловдив');
});
});

describe('userTexts (the redaction set for a multi-turn prompt)', () => {
const msgs = (parts: [string, string][]) =>
parts.map(([role, text]) => ({
role,
parts: [{ type: 'text', text }],
})) as unknown as Parameters<typeof userTexts>[0];

it('collects EVERY user turn, not just the latest', () => {
// A provider error can quote back any part of the prompt, and a multi-turn prompt carries the
// earlier questions too — redacting only ctx.userQuestion left those in the tail log (review f/u).
const earlier = 'колко плати община Пловдив през 2023';
const latest = 'а през 2024 колко плати същата община';
expect(
userTexts(
msgs([
['user', earlier],
['assistant', 'отговор'],
['user', latest],
]),
),
).toEqual([earlier, latest]);
});

it("ignores non-user roles and empty texts (they are not the person's words)", () => {
expect(
userTexts(
msgs([
['assistant', 'наш текст'],
['user', ' '],
['system', 'директива'],
]),
),
).toEqual([]);
});

it('feeds the logger a set that blanks an EARLIER question', () => {
const lines: string[] = [];
const spy = vi.spyOn(console, 'error').mockImplementation((l: unknown) => {
lines.push(String(l));
});
const earlier = 'колко плати община Пловдив през 2023';
const set = userTexts(
msgs([
['user', earlier],
['user', 'а през 2024 колко плати същата община'],
]),
);
makeStreamErrorLogger(set)(new Error(`400 invalid input: ${earlier}`));
spy.mockRestore();
expect(lines[0]).not.toContain('Пловдив');
expect(lines[0]).toContain('«редактирано»');
});
});
Loading
Loading