Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
251 commits
Select commit Hold shift + click to select a range
e8912ae
feat(frontend): align table row hover across sticky cells with BCDS
inderdeepsinghgill Jun 26, 2026
95ce487
fix(frontend): enforce sticky cell row hover highlighting in tables
inderdeepsinghgill Jun 26, 2026
906d4f2
docs: add extraction experiments suite design spec
strukalex May 8, 2026
98ddca8
docs(spec): replace generic checklist with codebase-derived; add exis…
strukalex May 8, 2026
dbb21f4
docs(spec): trim fat - drop duplicates, speculation, and pre-approval…
strukalex May 8, 2026
d636ea4
docs(spec): switch from gpt-5.5 to gpt-5 (quota-driven)
strukalex May 8, 2026
09d0045
docs(experiments): add brief library, hub doc, and provider architect…
strukalex May 8, 2026
edcb1a2
chore(experiments): env-template additions and data/datasets/ folder …
strukalex May 8, 2026
97e4942
feat(experiments): per-node Azure OpenAI deployment selection + local…
strukalex May 8, 2026
c3ae282
feat(seed): support flat-pair dataset drops, auto-derive manifest
strukalex May 8, 2026
e5668a5
docs(briefs): add gpt-5.5 quota-request prework block to E04 brief
strukalex May 8, 2026
b3c84c3
feat(seed): sync local datasets to blob storage on backend startup (d…
strukalex May 8, 2026
7afc746
fix(seed): use prismaService.prisma client, not the wrapper service i…
strukalex May 8, 2026
2cd3c7a
feat(experiments): make benchmarks invokable end-to-end after E01-E05…
strukalex May 8, 2026
71df3e3
fix(experiments): align with codebase workflow rules; benchmark on fu…
strukalex May 8, 2026
f9c3125
feat(seed): auto-discover experiment workflow templates from JSON
strukalex May 8, 2026
1dbfc44
feat(experiments): chained-branch stack + simplified E01 + comparison…
strukalex May 8, 2026
cb42a45
fix(briefs): E01 neural model id is "test-model" (not "test")
strukalex May 8, 2026
13f7187
fix(briefs): E01 neural model id is sdpr_synth_test (already the defa…
strukalex May 8, 2026
7639187
feat(experiment-01): wire neural DI workflow + post-processing chain
strukalex May 8, 2026
872267b
fix(temporal): make `npm run dev` reload reliably + archive E01 bench…
strukalex May 8, 2026
e919ce6
chore(experiment-01): runtime workflow tests + shared-rules updates
strukalex May 9, 2026
275dd74
chore(experiment-01): skip runtime workflow tests in CI
strukalex May 9, 2026
64dc11c
docs(briefs): drop the APIM-vs-direct DI TODO from E01
strukalex May 9, 2026
7e50fcb
docs(briefs): close gaps in E02 brief before next-chat handoff
strukalex May 9, 2026
40a9625
feat(experiment-02): Mistral Document AI on Azure AI Foundry provider
strukalex May 9, 2026
dc962ee
fix(experiment-02): make Mistral Foundry annotation actually run
strukalex May 9, 2026
265f087
feat(experiment-02): tune extraction prompts; E02 ≈ E01 on this dataset
strukalex May 9, 2026
033f77c
feat(experiment-02): force-resync mode + canonical run on aligned 40-…
strukalex May 9, 2026
25745f3
docs(briefs): bake E02 lessons into _shared-rules.md and E03 brief
strukalex May 10, 2026
e58cb17
feat(experiment-03): Azure Content Understanding provider + canonical…
strukalex May 10, 2026
0cdd90e
docs(experiment-03): meta-process retrospective + E04 startup prompt
strukalex May 10, 2026
6ba2581
docs(experiment-03): swap gpt-5.5 -> gpt-5.4 in E04 prompt (no quota …
strukalex May 10, 2026
2c18877
docs(experiment-03): scope E04 prompt to single-pass + gpt-5.4 only
strukalex May 10, 2026
f592545
docs(experiment-03): drop pdf.renderToImages from E04 prompt; add run…
strukalex May 10, 2026
902a1d6
feat(experiment-04): VLM-direct provider (gpt-5.4) + canonical 40-sam…
strukalex May 10, 2026
c5488b3
docs(experiment-04): correct iteration-vs-benchmark framing in SUMMARY
strukalex May 10, 2026
b17f39f
docs(experiments): add POST_BENCHMARK_FOLLOWUPS for cross-experiment …
strukalex May 10, 2026
224f9ad
feat(experiment-05): VLM + OCR hybrid provider (gpt-5.4) + canonical …
strukalex May 10, 2026
a4cee8e
improve(eval+e02): switch evaluator to strict + format-preservation p…
strukalex May 10, 2026
093edf0
improve(eval+e02): one-of GT support + round-2 Mistral prompt (conver…
strukalex May 10, 2026
7b525e7
tools: dump-errors-for-gt-cleanup + promote-gt-format-variants scripts
strukalex May 10, 2026
03089e5
chore(dataset): move samples-mix/private → public; update all references
strukalex May 10, 2026
d526ebd
dataset: promote sin/date/phone GT to one-of arrays where engine read…
strukalex May 10, 2026
56f9f75
dataset: expand sin/spouse_sin GT to three canonical 9-digit variants
strukalex May 10, 2026
35d3261
improve(e02): plumb OCR-3 features as opt-in; Foundry doesn't honor them
strukalex May 10, 2026
a18f9bf
improve(e02): re-benchmark on cleaned GT — new canonical (pass_rate 0…
strukalex May 10, 2026
eb8e6bb
improve(e02): probe naming-asymmetry + circle-checkbox prompt; both r…
strukalex May 11, 2026
ca9beb6
docs(improve-02-brief): prompt for the next improve branch — E03/E04/…
strukalex May 11, 2026
3e5ea14
improve(e03/e04/e05): strict re-eval on cleaned GT — all three engine…
strukalex May 11, 2026
2d37f1a
e06: cross-engine comparison report + ensemble combiner (E00/E02-E05)
strukalex May 11, 2026
513cd59
report: move cross-engine comparison to results/report/ + embed plots…
strukalex May 11, 2026
3990e8f
report: major rewrite per user feedback — metric descriptions, per-en…
strukalex May 11, 2026
1445c97
evaluator: fix FP/FN to match standard OCR-extraction definitions, re…
strukalex May 12, 2026
b49963c
report: rewrite intro + per-category + heatmap sections for external …
strukalex May 12, 2026
0277510
report: second pass of external-facing rewrites — add methodology, re…
strukalex May 13, 2026
2214245
feat(experiment-07): VLM + OCR hybrid with gpt-4o (model comparison v…
strukalex May 13, 2026
a77e0a5
feat(experiment-08): VLM + OCR hybrid with gpt-5.2 (model bake-off wi…
strukalex May 13, 2026
cc2716a
report: add E07 (gpt-4o) + E08 (gpt-5.2) to cross-engine comparison
strukalex May 13, 2026
e2ef2e4
report: restore E06 to charts/data + add E05 replication-check note
strukalex May 13, 2026
cbd538c
report: E03 re-run + mechanism analysis of E08 vs E03 gap
strukalex May 13, 2026
9e1ab87
report: external-facing rewrites — intro, CU architecture, E01, date …
strukalex May 14, 2026
633d6e8
report: second pass of external-facing rewrites — add methodology, re…
strukalex May 14, 2026
0e419ad
report: regenerate REPORT.pdf to match the latest REPORT.md
strukalex May 14, 2026
f6d2c5b
report: add exec summary, cost-per-page section, and form-under-test …
strukalex May 15, 2026
850b5e9
report: add Appendix B with verbatim extraction prompts
strukalex May 15, 2026
c6f7b14
report: tighten Cost section; add document-class framing and Canada P…
strukalex May 16, 2026
f8d730d
e01: re-run on production neural model — SUMMARY rewrite + GT-cleanup…
strukalex May 16, 2026
b17c35f
e01: re-evaluate against current GT + promote currency-prefix variants
strukalex May 16, 2026
11e7fef
cross-experiment: promote numeric/text-equivalence GT variants + unif…
strukalex May 16, 2026
07c4401
docs: before/after metric comparison for the GT-cleanup session
strukalex May 16, 2026
3b811d2
cross-experiment: extend numeric promotion to handle commas + interna…
strukalex May 16, 2026
2de4615
evaluator + promote: signature presence-only, :garbled: wildcard, new…
strukalex May 16, 2026
8063c24
report: refresh all metrics, plots, and CSVs against re-evaluated ben…
strukalex May 16, 2026
433f7d5
report: add E01 (Azure DI Neural custom model) to the comparison
strukalex May 16, 2026
583ed77
report: rewrite Reflection #9 to match updated evaluator behaviour
strukalex May 16, 2026
544bfc2
dataset: remove stray case_id field from 81 blank / 81 coffee GT
strukalex May 16, 2026
20499ee
report: fix stale per-engine claims after metric refresh
strukalex May 16, 2026
01c9596
ensemble: rebuild E06 across all 8 engines, switch to per-category sp…
strukalex May 16, 2026
92798a8
benchmark analysis: add cross-engine comparison, normaliser, audit re…
strukalex May 17, 2026
f47f1aa
scripts: add oc-export-benchmark-ocr-cache.sh for streaming benchmark
strukalex May 18, 2026
f82490a
benchmark analysis: standalone numeric-zero recovery + diagnostic suite
strukalex May 18, 2026
2c40def
recover-numeric-zeros: add row-label + positional fallback table finders
strukalex May 18, 2026
6bfd571
inspect-missing-zeros: mirror the recovery activity's A+B locator
strukalex May 18, 2026
eac7fed
recover-numeric-zeros: accept cells where stripped content already pa…
strukalex May 18, 2026
1d13cb1
benchmark normaliser: fuzzy text matching, single-digit-to-zero rule,…
strukalex May 18, 2026
4a5b960
benchmark analysis: add HITL capacity planner (target-recall sweep)
strukalex May 18, 2026
097d4e6
report-errors: per-occurrence wrong-by-category with baseline context
strukalex May 18, 2026
c98470c
hitl-planner: --exclude-missing-in-categories flag; compare-engines p…
strukalex May 18, 2026
ff92e78
hitl-planner: reviewable-cell workload metric; add benchmark pipeline…
strukalex May 18, 2026
3558ac6
docs: move benchmark-analysis pipeline doc into scripts/benchmark ana…
strukalex May 18, 2026
9232f13
hitl-planner: --skip-trivial-predictions-in-categories flag
strukalex May 18, 2026
7514ea5
pipeline doc: document the manual-annotation workflow
strukalex May 19, 2026
a636fac
md-to-pdf: avoid splitting plots across pages; gitignore analysis scr…
strukalex May 19, 2026
7558408
report: add stat-significance section; soften E08 framing; correct E0…
strukalex May 20, 2026
34bac0b
report: shorten production-path section; add industry-context subsection
strukalex May 20, 2026
30d4c13
report: strip speculation, dedupe per-engine section, drop failure-mo…
strukalex May 20, 2026
79c1626
report: strip internal source-code links; fix tilde strikethrough and…
strukalex May 22, 2026
0215f0a
benchmark analysis: HITL pass-through, all-predictions audit, per-fie…
strukalex May 28, 2026
d7bf989
test(experiments): port per-workflow tests to develop GraphWorkflow API
alex-struk Jun 27, 2026
1edd746
docs: extraction-experiments PR review + issue tracker
alex-struk Jun 29, 2026
3538d53
fix(vlm-direct): don't penalise confidence for genuinely-blank fields…
alex-struk Jun 29, 2026
3cfcd23
fix(vlm-ocr-hybrid): drop DI word confidence so HITL gate reflects ev…
alex-struk Jun 29, 2026
7c44f17
fix(azure-cu): handle inline-200 result shapes + fail fast on missing…
alex-struk Jun 29, 2026
198005f
fix(vlm): fenced-JSON parser ignores trailing content; dedupe parser …
alex-struk Jun 29, 2026
c23816f
fix(azure-di-read-plain): fail non-retryably on terminal analysis fai…
alex-struk Jun 29, 2026
8eaa91f
fix(azure-cu): omit valueDate for blank date fields (B6)
alex-struk Jun 29, 2026
584f484
fix(evaluator): structural comparison for nested objects + table rows…
alex-struk Jun 29, 2026
9daed44
refactor(evaluator): currency-aware numeric parse + shared Levenshtei…
alex-struk Jun 29, 2026
9fe860a
fix(vlm): rethrow DB errors in loadTemplate instead of masking as 'no…
alex-struk Jun 29, 2026
5e69b0d
refactor(azure-cu): hoist readEnv/sleep into azure-cu-client (R3)
alex-struk Jun 29, 2026
67cc2b6
test(temporal): complete activity-registry fixture + exact bijection …
alex-struk Jun 29, 2026
9e7988a
test(azure-cu): unit-test inline-200 result handling; close T2/T5 (T2…
alex-struk Jun 29, 2026
fe25e5a
docs: close U1 (redactCtxForQuery fix present on branch, lands with t…
alex-struk Jun 29, 2026
9d90f68
refactor(vlm-hybrid): accept OcrPayloadRef for layoutResponse (R1, st…
alex-struk Jun 30, 2026
272259f
docs: continuation/handoff plan for R1/R2/harness/T4/T6 refactor
alex-struk Jun 30, 2026
9bac736
refactor(e05): replace azure-di-read-plain with standard DI submit/po…
alex-struk Jun 30, 2026
b8e0809
refactor(mistral): merge native + Foundry activities into one variant…
alex-struk Jun 30, 2026
6d3fd12
test(e04): convert VLM-direct runtime to mock-only-paid harness; fix T4
alex-struk Jun 30, 2026
4014535
test(e01): convert neural-DI runtime to mock-only-paid harness
alex-struk Jun 30, 2026
285b56b
test(e03): convert Content Understanding runtime to mock-only-paid ha…
alex-struk Jun 30, 2026
3840c77
hitl: inline canvas-overlay editor, confidence-tier colors, smarter a…
strukalex Jun 1, 2026
fa6276d
fix(hitl): per-field key on CanvasFieldOverlay restores Tab focus (H1…
alex-struk Jun 29, 2026
2d0b75d
fix(hitl): shared text-measure module + rotation guard for overlay (H…
alex-struk Jun 29, 2026
31862fc
docs: add extraction experiments suite design spec
strukalex May 8, 2026
557f158
docs(spec): replace generic checklist with codebase-derived; add exis…
strukalex May 8, 2026
bea83b5
docs(spec): trim fat - drop duplicates, speculation, and pre-approval…
strukalex May 8, 2026
8722185
docs(spec): switch from gpt-5.5 to gpt-5 (quota-driven)
strukalex May 8, 2026
35f8d92
docs(experiments): add brief library, hub doc, and provider architect…
strukalex May 8, 2026
1300102
chore(experiments): env-template additions and data/datasets/ folder …
strukalex May 8, 2026
d3da3da
feat(experiments): per-node Azure OpenAI deployment selection + local…
strukalex May 8, 2026
0890334
feat(seed): support flat-pair dataset drops, auto-derive manifest
strukalex May 8, 2026
97d65c7
docs(briefs): add gpt-5.5 quota-request prework block to E04 brief
strukalex May 8, 2026
1206a7f
feat(seed): sync local datasets to blob storage on backend startup (d…
strukalex May 8, 2026
b907530
fix(seed): use prismaService.prisma client, not the wrapper service i…
strukalex May 8, 2026
ed6db2c
feat(experiments): make benchmarks invokable end-to-end after E01-E05…
strukalex May 8, 2026
1f1bad0
fix(experiments): align with codebase workflow rules; benchmark on fu…
strukalex May 8, 2026
2e26b72
feat(seed): auto-discover experiment workflow templates from JSON
strukalex May 8, 2026
9cdb707
feat(experiments): chained-branch stack + simplified E01 + comparison…
strukalex May 8, 2026
1e59c01
fix(briefs): E01 neural model id is "test-model" (not "test")
strukalex May 8, 2026
75a891a
fix(briefs): E01 neural model id is sdpr_synth_test (already the defa…
strukalex May 8, 2026
0fa6d97
docs(seed): document local-folder dataset seeding (parser + seed + bl…
alex-struk Jul 2, 2026
7575ca8
feat(experiment-01): wire neural DI workflow + post-processing chain
strukalex May 8, 2026
9e0e41a
fix(temporal): make `npm run dev` reload reliably + archive E01 bench…
strukalex May 8, 2026
154fb4c
chore(experiment-01): runtime workflow tests + shared-rules updates
strukalex May 9, 2026
b03a8d2
chore(experiment-01): skip runtime workflow tests in CI
strukalex May 9, 2026
0df9b12
docs(briefs): drop the APIM-vs-direct DI TODO from E01
strukalex May 9, 2026
56c4671
docs(briefs): close gaps in E02 brief before next-chat handoff
strukalex May 9, 2026
cf432e6
feat(experiment-02): Mistral Document AI on Azure AI Foundry provider
strukalex May 9, 2026
cb08164
fix(experiment-02): make Mistral Foundry annotation actually run
strukalex May 9, 2026
29c979a
feat(experiment-02): tune extraction prompts; E02 ≈ E01 on this dataset
strukalex May 9, 2026
e4540e8
feat(experiment-02): force-resync mode + canonical run on aligned 40-…
strukalex May 9, 2026
e37ea13
docs(briefs): bake E02 lessons into _shared-rules.md and E03 brief
strukalex May 10, 2026
7640341
feat(experiment-03): Azure Content Understanding provider + canonical…
strukalex May 10, 2026
df89976
docs(experiment-03): meta-process retrospective + E04 startup prompt
strukalex May 10, 2026
06b9f41
docs(experiment-03): swap gpt-5.5 -> gpt-5.4 in E04 prompt (no quota …
strukalex May 10, 2026
2ac3dae
docs(experiment-03): scope E04 prompt to single-pass + gpt-5.4 only
strukalex May 10, 2026
f96096d
docs(experiment-03): drop pdf.renderToImages from E04 prompt; add run…
strukalex May 10, 2026
21cc3ab
feat(experiment-04): VLM-direct provider (gpt-5.4) + canonical 40-sam…
strukalex May 10, 2026
a96b075
docs(experiment-04): correct iteration-vs-benchmark framing in SUMMARY
strukalex May 10, 2026
13beb03
docs(experiments): add POST_BENCHMARK_FOLLOWUPS for cross-experiment …
strukalex May 10, 2026
7454edb
feat(experiment-05): VLM + OCR hybrid provider (gpt-5.4) + canonical …
strukalex May 10, 2026
158a608
improve(eval+e02): switch evaluator to strict + format-preservation p…
strukalex May 10, 2026
41352d4
improve(eval+e02): one-of GT support + round-2 Mistral prompt (conver…
strukalex May 10, 2026
40a5e35
tools: dump-errors-for-gt-cleanup + promote-gt-format-variants scripts
strukalex May 10, 2026
0cf3126
chore(dataset): move samples-mix/private → public; update all references
strukalex May 10, 2026
e5e3616
dataset: promote sin/date/phone GT to one-of arrays where engine read…
strukalex May 10, 2026
2a28a91
dataset: expand sin/spouse_sin GT to three canonical 9-digit variants
strukalex May 10, 2026
9e46c6d
improve(e02): plumb OCR-3 features as opt-in; Foundry doesn't honor them
strukalex May 10, 2026
ea5fbba
improve(e02): re-benchmark on cleaned GT — new canonical (pass_rate 0…
strukalex May 10, 2026
4758c7c
improve(e02): probe naming-asymmetry + circle-checkbox prompt; both r…
strukalex May 11, 2026
d8908f2
docs(improve-02-brief): prompt for the next improve branch — E03/E04/…
strukalex May 11, 2026
fed6723
improve(e03/e04/e05): strict re-eval on cleaned GT — all three engine…
strukalex May 11, 2026
1449cca
e06: cross-engine comparison report + ensemble combiner (E00/E02-E05)
strukalex May 11, 2026
2284be2
report: move cross-engine comparison to results/report/ + embed plots…
strukalex May 11, 2026
b3e1463
report: major rewrite per user feedback — metric descriptions, per-en…
strukalex May 11, 2026
06dfdcb
evaluator: fix FP/FN to match standard OCR-extraction definitions, re…
strukalex May 12, 2026
0bef389
report: rewrite intro + per-category + heatmap sections for external …
strukalex May 12, 2026
a219c5b
report: second pass of external-facing rewrites — add methodology, re…
strukalex May 13, 2026
d8f7519
feat(experiment-07): VLM + OCR hybrid with gpt-4o (model comparison v…
strukalex May 13, 2026
3b18504
feat(experiment-08): VLM + OCR hybrid with gpt-5.2 (model bake-off wi…
strukalex May 13, 2026
b7ef9c4
report: add E07 (gpt-4o) + E08 (gpt-5.2) to cross-engine comparison
strukalex May 13, 2026
2d6b66c
report: restore E06 to charts/data + add E05 replication-check note
strukalex May 13, 2026
9f3b0a0
report: E03 re-run + mechanism analysis of E08 vs E03 gap
strukalex May 13, 2026
fa8860b
report: external-facing rewrites — intro, CU architecture, E01, date …
strukalex May 14, 2026
351f9d7
report: second pass of external-facing rewrites — add methodology, re…
strukalex May 14, 2026
fde83c0
report: regenerate REPORT.pdf to match the latest REPORT.md
strukalex May 14, 2026
4c13d22
report: add exec summary, cost-per-page section, and form-under-test …
strukalex May 15, 2026
12467a5
report: add Appendix B with verbatim extraction prompts
strukalex May 15, 2026
cf917e8
report: tighten Cost section; add document-class framing and Canada P…
strukalex May 16, 2026
6164ed0
e01: re-run on production neural model — SUMMARY rewrite + GT-cleanup…
strukalex May 16, 2026
ec00ace
e01: re-evaluate against current GT + promote currency-prefix variants
strukalex May 16, 2026
822d0c5
cross-experiment: promote numeric/text-equivalence GT variants + unif…
strukalex May 16, 2026
d4e9019
docs: before/after metric comparison for the GT-cleanup session
strukalex May 16, 2026
c0d89e1
cross-experiment: extend numeric promotion to handle commas + interna…
strukalex May 16, 2026
c7c12ec
evaluator + promote: signature presence-only, :garbled: wildcard, new…
strukalex May 16, 2026
c3ebdb3
report: refresh all metrics, plots, and CSVs against re-evaluated ben…
strukalex May 16, 2026
fe0d29d
report: add E01 (Azure DI Neural custom model) to the comparison
strukalex May 16, 2026
7925457
report: rewrite Reflection #9 to match updated evaluator behaviour
strukalex May 16, 2026
507c2ff
dataset: remove stray case_id field from 81 blank / 81 coffee GT
strukalex May 16, 2026
bd44c10
report: fix stale per-engine claims after metric refresh
strukalex May 16, 2026
bda6942
ensemble: rebuild E06 across all 8 engines, switch to per-category sp…
strukalex May 16, 2026
9498392
benchmark analysis: add cross-engine comparison, normaliser, audit re…
strukalex May 17, 2026
18157a1
scripts: add oc-export-benchmark-ocr-cache.sh for streaming benchmark
strukalex May 18, 2026
24d5d35
benchmark analysis: standalone numeric-zero recovery + diagnostic suite
strukalex May 18, 2026
89075be
recover-numeric-zeros: add row-label + positional fallback table finders
strukalex May 18, 2026
e4f178b
inspect-missing-zeros: mirror the recovery activity's A+B locator
strukalex May 18, 2026
77cae84
recover-numeric-zeros: accept cells where stripped content already pa…
strukalex May 18, 2026
f8f9f93
benchmark normaliser: fuzzy text matching, single-digit-to-zero rule,…
strukalex May 18, 2026
c884d64
benchmark analysis: add HITL capacity planner (target-recall sweep)
strukalex May 18, 2026
149999c
report-errors: per-occurrence wrong-by-category with baseline context
strukalex May 18, 2026
12b0268
hitl-planner: --exclude-missing-in-categories flag; compare-engines p…
strukalex May 18, 2026
4577bdc
hitl-planner: reviewable-cell workload metric; add benchmark pipeline…
strukalex May 18, 2026
6260a48
docs: move benchmark-analysis pipeline doc into scripts/benchmark ana…
strukalex May 18, 2026
6cd591e
hitl-planner: --skip-trivial-predictions-in-categories flag
strukalex May 18, 2026
27bc869
pipeline doc: document the manual-annotation workflow
strukalex May 19, 2026
139dc7b
md-to-pdf: avoid splitting plots across pages; gitignore analysis scr…
strukalex May 19, 2026
1b26559
report: add stat-significance section; soften E08 framing; correct E0…
strukalex May 20, 2026
6674d1e
report: shorten production-path section; add industry-context subsection
strukalex May 20, 2026
b88927a
report: strip speculation, dedupe per-engine section, drop failure-mo…
strukalex May 20, 2026
3f0aeb6
report: strip internal source-code links; fix tilde strikethrough and…
strukalex May 22, 2026
e984086
benchmark analysis: HITL pass-through, all-predictions audit, per-fie…
strukalex May 28, 2026
2de2e48
test(experiments): port per-workflow tests to develop GraphWorkflow API
alex-struk Jun 27, 2026
b05db1e
docs: extraction-experiments PR review + issue tracker
alex-struk Jun 29, 2026
9c6c28d
fix(vlm-direct): don't penalise confidence for genuinely-blank fields…
alex-struk Jun 29, 2026
00e09d7
fix(vlm-ocr-hybrid): drop DI word confidence so HITL gate reflects ev…
alex-struk Jun 29, 2026
cfef45d
fix(azure-cu): handle inline-200 result shapes + fail fast on missing…
alex-struk Jun 29, 2026
aadaa84
fix(vlm): fenced-JSON parser ignores trailing content; dedupe parser …
alex-struk Jun 29, 2026
5f22a2a
fix(azure-di-read-plain): fail non-retryably on terminal analysis fai…
alex-struk Jun 29, 2026
eb8955e
fix(azure-cu): omit valueDate for blank date fields (B6)
alex-struk Jun 29, 2026
05a3f69
fix(evaluator): structural comparison for nested objects + table rows…
alex-struk Jun 29, 2026
19f2172
refactor(evaluator): currency-aware numeric parse + shared Levenshtei…
alex-struk Jun 29, 2026
4dfff70
fix(vlm): rethrow DB errors in loadTemplate instead of masking as 'no…
alex-struk Jun 29, 2026
b89bf63
refactor(azure-cu): hoist readEnv/sleep into azure-cu-client (R3)
alex-struk Jun 29, 2026
2c8ea3e
test(temporal): complete activity-registry fixture + exact bijection …
alex-struk Jun 29, 2026
cb4a57e
test(azure-cu): unit-test inline-200 result handling; close T2/T5 (T2…
alex-struk Jun 29, 2026
4455d07
docs: close U1 (redactCtxForQuery fix present on branch, lands with t…
alex-struk Jun 29, 2026
3c72e73
refactor(vlm-hybrid): accept OcrPayloadRef for layoutResponse (R1, st…
alex-struk Jun 30, 2026
21d3cb4
docs: continuation/handoff plan for R1/R2/harness/T4/T6 refactor
alex-struk Jun 30, 2026
46d5931
refactor(e05): replace azure-di-read-plain with standard DI submit/po…
alex-struk Jun 30, 2026
ab33037
refactor(mistral): merge native + Foundry activities into one variant…
alex-struk Jun 30, 2026
602d99d
test(e04): convert VLM-direct runtime to mock-only-paid harness; fix T4
alex-struk Jun 30, 2026
4aad09f
test(e01): convert neural-DI runtime to mock-only-paid harness
alex-struk Jun 30, 2026
1f874d8
test(e03): convert Content Understanding runtime to mock-only-paid ha…
alex-struk Jun 30, 2026
e403611
refactor(scripts): move experiment tooling out of worker src into app…
alex-struk Jul 2, 2026
a80a509
refactor(evaluator): make presence-only matching a config-driven rule…
alex-struk Jul 2, 2026
634a0b0
docs: address PR #155 review — root .env note + rename briefs → integ…
alex-struk Jul 10, 2026
2fe1649
Merge remote-tracking branch 'origin/develop' into experiment/09-sdpr…
alex-struk Jul 13, 2026
f484f0a
refactor(evaluator): make presence-only matching a config-driven rule…
alex-struk Jul 2, 2026
d1d983b
Merge remote-tracking branch 'origin/experiment/08-part-2' into exper…
alex-struk Jul 13, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
The table of contents is too big for display.
Diff view
Diff view
  •  
  •  
  •  
19 changes: 19 additions & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,10 @@ node_modules
.env
dist

# Python bytecode caches
__pycache__/
*.pyc

# SpecStory
.specstory/
/.claude/settings.loca.json
Expand Down Expand Up @@ -34,6 +38,21 @@ infra/*.tfstate
infra/*.tfstate.backup
infra/.terraform.lock.hcl

# Local benchmark dataset drops — private subset never committed.
# See docs/superpowers/specs/2026-05-08-extraction-experiments-design.md
data/datasets/*/private/**
!data/datasets/*/private/.gitkeep
!data/datasets/.gitkeep

# SDPR-report working files — drafts, regenerated plots, and HITL curves used
# during analysis. Lived on a network share; kept locally only as a scratch
# space. Don't commit them — they contain analysis outputs derived from
# private benchmark data.
/SDPR_OCR_Performance_Report*.md
/SDPR_OCR_Performance_Report*.pdf
/plots/
/hitl/

# Generated repo wiki HTML (built by docs/build.sh at docs deploy time)
/docs/wiki.html
/docs/wiki-*.html
71 changes: 71 additions & 0 deletions apps/backend-services/src/azure/azure-openai.controller.spec.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,71 @@
import type { ConfigService } from "@nestjs/config";
import { AzureOpenAiController } from "./azure-openai.controller";

describe("AzureOpenAiController", () => {
function makeController(
env: Record<string, string | undefined>,
): AzureOpenAiController {
const configService = {
get: <T>(key: string): T | undefined => env[key] as T | undefined,
} as unknown as ConfigService;
return new AzureOpenAiController(configService);
}

describe("getDeployments", () => {
it("parses AZURE_OPENAI_DEPLOYMENTS as comma-separated list", async () => {
const controller = makeController({
AZURE_OPENAI_DEPLOYMENTS: "gpt-4o,gpt-5",
});
const result = await controller.getDeployments();
expect(result.deployments).toEqual(["gpt-4o", "gpt-5"]);
});

it("trims whitespace and drops empty entries", async () => {
const controller = makeController({
AZURE_OPENAI_DEPLOYMENTS: " gpt-4o , , gpt-5 , ",
});
const result = await controller.getDeployments();
expect(result.deployments).toEqual(["gpt-4o", "gpt-5"]);
});

it("falls back to AZURE_OPENAI_DEPLOYMENT when DEPLOYMENTS is unset", async () => {
const controller = makeController({
AZURE_OPENAI_DEPLOYMENT: "gpt-4o",
});
const result = await controller.getDeployments();
expect(result.deployments).toEqual(["gpt-4o"]);
});

it("falls back to AZURE_OPENAI_DEPLOYMENT when DEPLOYMENTS is empty string", async () => {
const controller = makeController({
AZURE_OPENAI_DEPLOYMENTS: "",
AZURE_OPENAI_DEPLOYMENT: "gpt-4o",
});
const result = await controller.getDeployments();
expect(result.deployments).toEqual(["gpt-4o"]);
});

it("returns empty array when neither var is set", async () => {
const controller = makeController({});
const result = await controller.getDeployments();
expect(result.deployments).toEqual([]);
});

it("returns empty array when both vars are empty strings", async () => {
const controller = makeController({
AZURE_OPENAI_DEPLOYMENTS: "",
AZURE_OPENAI_DEPLOYMENT: "",
});
const result = await controller.getDeployments();
expect(result.deployments).toEqual([]);
});

it("preserves deployment order from the env var", async () => {
const controller = makeController({
AZURE_OPENAI_DEPLOYMENTS: "gpt-5,gpt-4o,gpt-4o-mini",
});
const result = await controller.getDeployments();
expect(result.deployments).toEqual(["gpt-5", "gpt-4o", "gpt-4o-mini"]);
});
});
});
50 changes: 50 additions & 0 deletions apps/backend-services/src/azure/azure-openai.controller.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,50 @@
import { Controller, Get } from "@nestjs/common";
import { ConfigService } from "@nestjs/config";
import {
ApiOkResponse,
ApiOperation,
ApiTags,
ApiUnauthorizedResponse,
} from "@nestjs/swagger";
import { Identity } from "@/auth/identity.decorator";
import { AzureOpenAiDeploymentsResponseDto } from "@/azure/dto/azure-openai-deployments-response.dto";

/**
* Lists Azure OpenAI deployments the workflow editor / activity nodes may select.
*
* Allow-list is sourced from AZURE_OPENAI_DEPLOYMENTS (comma-separated). When unset
* the controller falls back to the single deployment name in AZURE_OPENAI_DEPLOYMENT
* for backward compatibility, or an empty array if neither is set.
*/
@ApiTags("Azure OpenAI")
@Controller("api/azure-openai")
export class AzureOpenAiController {
constructor(private readonly configService: ConfigService) {}

@Get("deployments")
@Identity({ allowApiKey: true })
@ApiOperation({
summary:
"List Azure OpenAI deployments allowed for workflow node selection",
})
@ApiOkResponse({
description: "Allowed deployment names",
type: AzureOpenAiDeploymentsResponseDto,
})
@ApiUnauthorizedResponse({ description: "Caller is not authenticated." })
async getDeployments(): Promise<AzureOpenAiDeploymentsResponseDto> {
const list = this.configService.get<string>("AZURE_OPENAI_DEPLOYMENTS");
if (list && list.trim() !== "") {
const deployments = list
.split(",")
.map((s) => s.trim())
.filter((s) => s.length > 0);
return { deployments };
}

const fallback = this.configService.get<string>("AZURE_OPENAI_DEPLOYMENT");
return {
deployments: fallback && fallback.trim() !== "" ? [fallback.trim()] : [],
};
}
}
3 changes: 2 additions & 1 deletion apps/backend-services/src/azure/azure.module.ts
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
import { Module } from "@nestjs/common";
import { AzureController } from "@/azure/azure.controller";
import { AzureService } from "@/azure/azure.service";
import { AzureOpenAiController } from "@/azure/azure-openai.controller";
import { ClassifierService } from "@/azure/classifier.service";
import { ClassifierDbService } from "@/azure/classifier-db.service";
import { ClassifierOrphanCleanupService } from "@/azure/classifier-orphan-cleanup.service";
Expand All @@ -17,6 +18,6 @@ import { BlobStorageModule } from "@/blob-storage/blob-storage.module";
],
exports: [AzureService],
imports: [BlobStorageModule],
controllers: [AzureController],
controllers: [AzureController, AzureOpenAiController],
})
export class AzureModule {}
Original file line number Diff line number Diff line change
@@ -0,0 +1,11 @@
import { ApiProperty } from "@nestjs/swagger";

export class AzureOpenAiDeploymentsResponseDto {
@ApiProperty({
description:
"Azure OpenAI deployment names allowed for selection on workflow nodes (e.g. for the enrichResults activity's azureOpenAiDeployment parameter). Parsed from the AZURE_OPENAI_DEPLOYMENTS env var (comma-separated). The first entry is treated as the default if a workflow node doesn't specify one.",
example: ["gpt-4o", "gpt-5"],
type: [String],
})
deployments!: string[];
}
2 changes: 2 additions & 0 deletions apps/backend-services/src/benchmark/benchmark.module.ts
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@ import { DatabaseModule } from "@/database/database.module";
import { DocumentModule } from "@/document/document.module";
import { HitlModule } from "@/hitl/hitl.module";
import { OcrModule } from "@/ocr/ocr.module";
import { LocalDatasetSyncService } from "@/seed/local-dataset-sync.service";
import { TemporalModule } from "@/temporal/temporal.module";
import { WorkflowModule } from "@/workflow/workflow.module";
import { AiRecommendationService } from "./ai-recommendation.service";
Expand Down Expand Up @@ -72,6 +73,7 @@ import { OcrImprovementPipelineService } from "./ocr-improvement-pipeline.servic
AiRecommendationService,
OcrImprovementPipelineService,
BenchmarkErrorDetectionService,
LocalDatasetSyncService,
],
exports: [
DatasetService,
Expand Down
Loading
Loading