Repository navigation
Expand file tree
/
Copy pathseed.py
More file actions
343 lines (321 loc) · 13.5 KB
/
Copy pathseed.py
File metadata and controls
343 lines (321 loc) · 13.5 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
"""Seed script — initializes the schema and inserts placeholder knowledge entries.
Run once before deploying the MCP reader for the first time:
uv run python seed.py
Idempotent: entries that already exist (matched by title) are skipped.
"""
from __future__ import annotations
import os
import duckdb
from db import _DDL, _seed_lookups # noqa: PLC2701
_DB_PATH = os.environ.get("ENTERPRISE_DB_PATH", "./enterprise.duckdb")
# ---------------------------------------------------------------------------
# Placeholder entries per category (title, content, category_name, roles, tags)
# ---------------------------------------------------------------------------
_ENTRIES: list[tuple[str, str, str, list[str], list[str]]] = [
# ── Development process ──────────────────────────────────────────────
(
"General Coding Standards",
(
"All code must be reviewed before merging. "
"Functions should do one thing. Avoid magic numbers — use named constants. "
"Write self-documenting code; add comments only when intent is not obvious. "
"Follow the project's established style guide."
),
"development_process",
["developer", "architect"],
["standards"],
),
(
"New Project Checklist",
(
"Before a new project goes live ensure: "
"1) Repository created and protected-branch rules applied. "
"2) CI/CD pipeline configured with security scanning. "
"3) README with setup, run, and deploy instructions. "
"4) Observability: logs, metrics, and health endpoint. "
"5) Production readiness sign-off from architecture team."
),
"development_process",
["developer", "devops", "architect"],
["new-project", "standards"],
),
(
"Dependency Management Policy",
(
"New third-party dependencies require approval via the dependency review process. "
"Pin all dependency versions in lockfiles (uv.lock, package-lock.json, etc.). "
"Run vulnerability scans on every build. "
"Deprecated or abandoned packages must be replaced within 90 days of notification."
),
"development_process",
["developer"],
["dependencies", "standards"],
),
# ── CI/CD ────────────────────────────────────────────────────────────
(
"CI/CD Pipeline Standards",
(
"Every repository must have an automated pipeline that runs on every push to main "
"and on every pull request. Required stages: build → unit tests → SAST scan → "
"DAST scan (non-prod) → quality gate → deploy to staging → smoke tests → "
"manual approval → deploy to production."
),
"cicd",
["developer", "devops"],
["pipeline"],
),
(
"Security Scanning Requirements",
(
"Veracode SAST must be run on every build targeting the main branch. "
"Critical and High findings block the pipeline. Medium findings require a "
"documented exception approved by the security team within 30 days. "
"SonarQube is run on every PR; the quality gate must pass before merge."
),
"cicd",
["developer", "devops"],
["security-scanning", "pipeline"],
),
(
"Quality Gate Requirements",
(
"Minimum code coverage: 80% overall, 70% per new file introduced in a PR. "
"SonarQube quality gate: 0 blocker issues, 0 critical issues, "
"duplication < 3%, maintainability rating A or B. "
"Coverage reports must be published as pipeline artefacts."
),
"cicd",
["developer"],
["quality-gates"],
),
(
"Artifact Management Rules",
(
"Build artefacts must be published to the company's internal registry. "
"Use semantic versioning (MAJOR.MINOR.PATCH). "
"Artefacts older than 180 days and not referenced by a live deployment are deleted. "
"Production images must be signed and verified before deployment."
),
"cicd",
["devops"],
["artifacts"],
),
# ── Security ─────────────────────────────────────────────────────────
(
"Authentication and Authorisation Policy",
(
"All APIs must use OAuth 2.0 / OIDC for authentication. "
"Authorisation must be enforced at the service layer — never trust client-side claims. "
"Tokens must be short-lived (≤1 hour). "
"Refresh tokens must be rotated on use and stored securely."
),
"security",
["developer", "architect"],
["security", "auth"],
),
(
"Compliance Requirements",
(
"All projects handling personal data must comply with GDPR. "
"Data classification must be performed before project kick-off. "
"PII must be pseudonymised or anonymised at rest. "
"Annual security audits are mandatory for Tier-1 systems."
),
"security",
["developer", "architect", "manager"],
["compliance"],
),
(
"Secrets Management Policy",
(
"Secrets must never be committed to source control. "
"Use the approved vault solution (HashiCorp Vault / cloud KMS) for all credentials. "
"Secrets are injected at runtime via environment variables or mounted volumes. "
"Rotate all secrets every 90 days. "
"Exposed credentials must be rotated within 4 hours of discovery."
),
"security",
["developer", "devops"],
["secrets", "security"],
),
# ── Production readiness ─────────────────────────────────────────────
(
"Production Readiness Checklist",
(
"A service is ready for production when: "
"1) All quality gates passed. "
"2) Runbook documented and reviewed. "
"3) On-call rotation configured. "
"4) Monitoring dashboards and alerts set up. "
"5) Load/stress test completed. "
"6) Rollback procedure documented and tested. "
"7) Architecture sign-off obtained."
),
"production_readiness",
["developer", "devops", "architect"],
["checklist"],
),
(
"Monitoring and Observability Standards",
(
"Every service must expose: "
"1) Structured JSON logs with correlation IDs. "
"2) /health endpoint returning HTTP 200 when healthy. "
"3) Prometheus-compatible /metrics endpoint. "
"4) Distributed tracing via OpenTelemetry. "
"SLO: 99.5% availability, p99 latency < 500 ms for Tier-1 services."
),
"production_readiness",
["developer", "devops"],
["monitoring"],
),
(
"Deployment Process",
(
"All production deployments must follow the blue-green or canary strategy. "
"Deployments require a 2-person approval in the deployment pipeline. "
"Deployment window: weekdays 10:00–14:00 UTC outside freeze periods. "
"Automated smoke tests must pass before traffic is shifted. "
"Rollback must be achievable within 10 minutes."
),
"production_readiness",
["devops"],
["deployment"],
),
(
"Incident Response Process",
(
"Severity levels: P1 (complete outage) → 15 min response; "
"P2 (degraded service) → 1 hour; P3 (minor issue) → next business day. "
"Declare incidents in the incident management tool. "
"War-room bridge opened for P1/P2. "
"Post-mortem required within 5 business days for P1 incidents."
),
"production_readiness",
["devops", "manager"],
["incidents"],
),
# ── Git / PR ─────────────────────────────────────────────────────────
(
"Pull Request Conventions",
(
"PR title format: `<type>(<scope>): <short description>` (Conventional Commits). "
"Description must include: motivation, what changed, and testing steps. "
"Link to the Jira/Linear ticket in the PR description. "
"Minimum 1 approving review required; 2 for changes to core libraries. "
"Merge strategy: squash-and-merge onto main."
),
"git_pr",
["developer"],
["pull-requests"],
),
(
"Branching Strategy",
(
"We follow trunk-based development with short-lived feature branches. "
"Branch naming: `<type>/<ticket-id>-<short-description>` "
"(e.g. `feat/PROJ-123-add-login`). "
"Feature branches must be merged within 5 business days. "
"Release branches: `release/vMAJOR.MINOR`. "
"Hotfixes branch from the release tag: `hotfix/vMAJOR.MINOR.PATCH`."
),
"git_pr",
["developer"],
["branching"],
),
(
"Code Review Standards",
(
"Reviewers must check: correctness, security implications, test coverage, "
"adherence to architecture guidelines, and documentation completeness. "
"Feedback must be constructive and specific. "
"Blocking comments must be resolved before merge. "
"Non-blocking suggestions are prefixed with `nit:` or `suggestion:`. "
"Reviews must be completed within 1 business day."
),
"git_pr",
["developer"],
["code-review"],
),
# ── Architecture ─────────────────────────────────────────────────────
(
"Architecture Guidelines",
(
"New systems must follow the approved domain-driven design boundaries. "
"Any decision that affects system topology, data persistence strategy, "
"or inter-service communication requires an Architecture Decision Record (ADR). "
"ADRs are reviewed by the architecture guild before implementation begins. "
"Prefer event-driven patterns for cross-domain communication."
),
"architecture",
["architect", "developer"],
["guidelines"],
),
(
"Technology Radar",
(
"ADOPT: Python (FastAPI), TypeScript (React), PostgreSQL, Kubernetes, OpenTelemetry. "
"TRIAL: DuckDB (analytics/edge), Temporal (workflows), Rust (performance-critical). "
"ASSESS: Wasm, eBPF, LLM-based agents. "
"HOLD: XML-based REST (SOAP), monolithic deployment units, "
"direct DB access from frontend services."
),
"architecture",
["architect", "developer", "manager"],
["tech-radar"],
),
]
def main() -> None:
print(f"Connecting to {_DB_PATH} ...")
con = duckdb.connect(_DB_PATH)
con.execute(_DDL)
_seed_lookups(con)
created = 0
skipped = 0
for title, content, category_name, role_names, tag_names in _ENTRIES:
existing = con.execute(
"SELECT id FROM knowledge_entries WHERE title = ?", [title]
).fetchone()
if existing:
print(f" SKIP {title!r}")
skipped += 1
continue
cat = con.execute(
"SELECT id FROM categories WHERE name = ?", [category_name]
).fetchone()
if not cat:
print(f" ERROR category not found: {category_name!r} — skipping {title!r}")
continue
row = con.execute(
"INSERT INTO knowledge_entries (title, content, category_id) VALUES (?, ?, ?) "
"RETURNING id",
[title, content, cat[0]],
).fetchone()
entry_id: int = row[0] # type: ignore[index]
for role_name in role_names:
role = con.execute("SELECT id FROM roles WHERE name = ?", [role_name]).fetchone()
if role:
con.execute(
"INSERT INTO entry_roles (entry_id, role_id) VALUES (?, ?) ON CONFLICT DO NOTHING",
[entry_id, role[0]],
)
for tag_name in tag_names:
tag = con.execute("SELECT id FROM tags WHERE name = ?", [tag_name]).fetchone()
if tag:
tag_id = tag[0]
else:
tag_row = con.execute(
"INSERT INTO tags (id, name) VALUES (nextval('tag_seq'), ?) RETURNING id",
[tag_name],
).fetchone()
tag_id = tag_row[0] # type: ignore[index]
con.execute(
"INSERT INTO entry_tags (entry_id, tag_id) VALUES (?, ?) ON CONFLICT DO NOTHING",
[entry_id, tag_id],
)
print(f" CREATE {title!r}")
created += 1
con.close()
print(f"\nDone — {created} created, {skipped} skipped.")
if __name__ == "__main__":
main()