commit d42f0c43924027dc3778c68fb25d53d4246ed40f Author: E.Gavrilov Date: Thu May 21 16:57:58 2026 +0300 Task 001: dockerized monorepo foundation diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 0000000..a58e026 --- /dev/null +++ b/.dockerignore @@ -0,0 +1,12 @@ +.git +.env +node_modules +**/node_modules +**/.next +__pycache__ +**/__pycache__ +.pytest_cache +.ruff_cache +.mypy_cache +dist +coverage diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..e420417 --- /dev/null +++ b/.env.example @@ -0,0 +1,21 @@ +# Local-only defaults for Docker Compose development. +# These values are intentionally non-production and must not be reused as real credentials. +COMPOSE_PROJECT_NAME=ai-content-pipeline + +FRONTEND_PORT=3000 +BACKEND_PORT=8000 +RUNNER_PORT=8010 + +POSTGRES_USER=pipeline +POSTGRES_PASSWORD=pipeline_local +POSTGRES_DB=pipeline +POSTGRES_PORT=5432 + +REDIS_PORT=6379 + +MINIO_ROOT_USER=minio_local +MINIO_ROOT_PASSWORD=minio_local_password +MINIO_API_PORT=9000 +MINIO_CONSOLE_PORT=9001 +OBJECT_STORAGE_BUCKET=pipeline-local +OBJECT_STORAGE_REGION=us-east-1 diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..291de22 --- /dev/null +++ b/.gitignore @@ -0,0 +1,11 @@ +.env +.DS_Store +node_modules/ +.next/ +__pycache__/ +*.pyc +.pytest_cache/ +.ruff_cache/ +.mypy_cache/ +dist/ +coverage/ diff --git a/IDEA.md b/IDEA.md new file mode 100644 index 0000000..2a5ac66 --- /dev/null +++ b/IDEA.md @@ -0,0 +1,2102 @@ +# Development specification: AI Content Pipeline v1 + +## 1. Objective + +Build an extendable content production pipeline for writers and editors. + +The system accepts a short article description, asks boundary-setting questions, generates an article plan, stops for editor review, then continues through research, evidence gathering, section scaffolding, visual generation, SEO, linguistic review, final approval, and Git-backed publishing into a Next.js content repository. + +v1 is an internal or single-tenant editorial tool. It is not a customer-facing multi-tenant SaaS product. + +Chosen architecture: + +```text +React / Next.js UI +FastAPI backend +LangGraph workflow engine +Postgres state store +Agent Runner Service using Codex CLI subscription authentication +Optional Claude Code runner for internal-only workflows +Object storage for files and assets +Git-backed publishing adapter layer for multiple target websites +``` + +LangGraph is appropriate because it supports durable execution, human-in-the-loop workflow control, and stateful orchestration. These are central to the required “stop, wait for edits, resume” behavior. ([docs.langchain.com][1]) + +Codex should be the primary subscription-based agent runner because Codex supports ChatGPT sign-in for subscription access, while Codex CLI also supports ChatGPT account login, API key login, or access-token login. For this project, use one managed runner identity per environment through ChatGPT subscription login or enterprise access-token auth, not usage-based API-key calls. ([OpenAI Разработчики][2]) + +Claude Code can be supported as an optional internal runner, but not as the main backend for routing third-party user jobs through consumer subscription credentials. Anthropic documents OAuth and API-key authentication as serving different purposes and states limits around third-party product/service usage. ([Claude Code][3]) + +--- + +# 2. Product scope + +## In scope for v1 + +```text +1. Article intake +2. Boundary question generation +3. Plan generation +4. Plan approval gate +5. Web research task +6. Evidence matrix generation +7. Parallel section scaffolding +8. Hero image prompt generation +9. Table and diagram specification generation +10. SEO pass +11. Linguistic and tone review +12. Final approval gate +13. Git-backed publish commit creation +14. Multi-site configuration +15. Job history and audit trail +16. Codex CLI-based agent execution +``` + +## Out of scope for v1 + +```text +1. Fully autonomous publishing without human approval +2. Multi-language localization +3. A/B testing +4. Content performance analytics +5. Legal/compliance review workflows +6. Real-time collaborative editing +7. Direct Anthropic/OpenAI API-token usage for generation +8. Complex media editing pipeline +9. Owning CI/CD deployment orchestration +10. Multi-tenant customer isolation +11. Per-user Codex runner credentials +``` + +The publishing step in v1 should create a **direct commit to the configured production branch** of a Git-backed Next.js site repository after final editor approval and a best-effort content-shape dry run. The target repository's existing CI/CD owns deployment. + +--- + +# 3. Core user roles + +## Admin + +Can configure target websites, pipeline parameters, agent runner profiles, prompt versions, publishing YAML, and site-specific transformation/upload scripts. + +## Editor + +Can create briefs, answer boundary questions, run the pipeline, edit intermediate results, approve plans, edit drafts, approve final content, and create publish commits. + +Only Admins can edit upload scripts and pipeline configuration. Admins are fully trusted code operators because admin-defined transformation scripts run directly on the runner host inside checked-out site repositories. + +--- + +# 4. High-level workflow + +```text +ARTICLE_BRIEF_CREATED + | + v +BOUNDARY_QUESTIONS_GENERATED + | + v +BOUNDARY_ANSWERS_SUBMITTED + | + v +PLAN_GENERATED + | + v +PLAN_REVIEW_REQUIRED + | + +--> editor requests changes --> PLAN_REVISION_REQUIRED --> PLAN_GENERATED + | + +--> editor approves + v +RESEARCH_RUNNING + | + v +EVIDENCE_MATRIX_READY + | + v +PARALLEL_PRODUCTION_RUNNING + | + +--> section scaffolds + +--> hero image prompt + +--> table specs + +--> diagram specs + +--> SEO brief + +--> FAQ block + | + v +DRAFT_ASSEMBLED + | + v +SEO_AND_LANGUAGE_REVIEW_READY + | + v +FINAL_REVIEW_REQUIRED + | + +--> editor requests changes --> FINAL_REVISION_REQUIRED + | + +--> editor approves + v +PUBLISH_DRY_RUN_REQUIRED + | + v +PUBLISH_COMMIT_READY + | + v +PUBLISH_COMMIT_CREATED + | + v +DONE +``` + +--- + +# 5. System architecture + +```text +Frontend + React / Next.js + | + v +Backend API + FastAPI + | + +--> Auth service + +--> Article service + +--> Site config service + +--> Workflow service + +--> Review service + +--> Asset service + +--> Publishing service + | + v +Workflow engine + LangGraph + | + v +Queue + Redis Queue / Celery / Dramatiq + | + v +Agent Runner Service + | + +--> Codex CLI runner + +--> Optional Claude Code runner + +--> Web research script + +--> Diagram script + +--> Git-backed publishing script + | + v +Storage + +--> Postgres + +--> Object storage + +--> Optional vector store +``` + +FastAPI’s built-in background tasks are useful for small post-response actions, but this system should use a real queue for long-running agent jobs, because article generation, research, retries, and approval waits are workflow-level tasks rather than short request-level tasks. FastAPI supports background tasks, but the durable workflow should live in LangGraph plus a queue. ([fastapi.tiangolo.com][4]) + +--- + +# 6. Main components + +## 6.1 Frontend UI + +### Required screens + +```text +1. Dashboard +2. New article brief +3. Boundary questions +4. Plan review +5. Research pack view +6. Evidence matrix view +7. Draft editor +8. Asset review +9. SEO review +10. Final approval +11. Publishing settings +12. Site configuration +13. Workflow history +14. Media library +15. Generic Markdown/MDX rich preview +``` + +v1 frontend should feel like a full editorial workspace, including rich preview, media management, and advanced draft editing. If timeline forces scope reduction, scheduling slips before rich preview. + +### Dashboard fields + +```text +Article title +Target website +Workflow status +Assigned editor +Last updated +Next required action +Publishing status +``` + +### Plan review UI must support + +```text +Approve plan +Request revision +Edit plan directly +Add section-level notes +Add source requirements +Add excluded sources +Add visual requirements +Change tone +Change target audience +Change SEO keyword +``` + +### Final review UI must support + +```text +Approve final draft +Request revision +Edit title +Edit meta description +Edit article body +Edit images +Edit tables +Edit diagrams +Edit internal links +Edit frontmatter, content path, category, and tags +Create publish commit +``` + +--- + +## 6.2 Backend API + +Backend stack: + +```text +FastAPI +Postgres +SQLAlchemy or SQLModel +Pydantic models +Redis Queue / Celery / Dramatiq +LangGraph workflow module +Object storage SDK +Git publishing adapter interfaces +``` + +### API design principles + +```text +1. Every workflow-changing action must be explicit. +2. Every human approval must be recorded. +3. Every agent output must be validated before entering the article state. +4. Every generated claim must be traceable to evidence or marked unsupported. +5. Every target website must be driven by config, not hardcoded logic. +6. Article/domain tables are authoritative for product state; LangGraph checkpoints are execution state only. +7. Every approval-relevant plan or draft edit creates a new immutable version. +``` + +--- + +# 7. Data model + +## 7.1 Article + +```json +{ + "id": "uuid", + "target_site_id": "uuid", + "created_by": "uuid", + "assigned_editor_id": "uuid", + "status": "PLAN_REVIEW_REQUIRED", + "brief_description": "string", + "working_title": "string", + "language": "en", + "content_type": "longform_guide", + "primary_keyword": "string", + "created_at": "datetime", + "updated_at": "datetime" +} +``` + +## 7.2 TargetSite + +```json +{ + "id": "uuid", + "name": "B2B SaaS Blog", + "slug": "b2b_saas_blog", + "publishing_type": "git_next", + "default_language": "en", + "brand_voice": "direct, useful, evidence-backed", + "audience": "B2B SaaS founders and content leads", + "seo_rules": {}, + "visual_rules": {}, + "source_rules": {}, + "publishing_rules": { + "repository_url": "git@github.com:example/site.git", + "production_branch": "main", + "content_format": "mdx", + "content_path_template": "content/articles/{slug}.mdx", + "asset_path_template": "public/articles/{slug}/{filename}", + "frontmatter_mapping": {}, + "transform_script_version_id": "uuid" + }, + "created_at": "datetime", + "updated_at": "datetime" +} +``` + +## 7.3 BoundaryQuestion + +```json +{ + "id": "uuid", + "article_id": "uuid", + "question": "Who is the article for?", + "question_type": "audience", + "answer": "Content operations leads", + "required": true, + "sort_order": 1 +} +``` + +## 7.4 ArticlePlan + +```json +{ + "id": "uuid", + "article_id": "uuid", + "version": 1, + "status": "PENDING_REVIEW", + "title_options": [], + "recommended_title": "string", + "search_intent": "string", + "thesis": "string", + "sections": [], + "claims_to_prove": [], + "visuals_needed": [], + "seo_notes": {}, + "risks": [], + "editor_notes": [], + "created_at": "datetime" +} +``` + +## 7.5 PlanSection + +```json +{ + "id": "uuid", + "article_plan_id": "uuid", + "sort_order": 1, + "heading": "string", + "purpose": "string", + "key_points": [], + "claims_to_support": [], + "evidence_needed": [], + "visuals_needed": [], + "target_word_count": 500 +} +``` + +## 7.6 EvidenceItem + +```json +{ + "id": "uuid", + "article_id": "uuid", + "source_title": "string", + "source_url": "string", + "source_domain": "string", + "source_type": "official_docs | research | news | company_blog | other", + "source_quality_score": 0.9, + "summary": "string", + "relevant_quotes": [], + "supports_claims": [], + "used_in_sections": [], + "artifact_manifest_id": "uuid", + "snapshot_object_key": "s3://bucket/research-runs/run_123/source_001.json", + "snapshot_content_hash": "sha256", + "retrieved_at": "datetime" +} +``` + +## 7.7 Claim + +```json +{ + "id": "uuid", + "article_id": "uuid", + "section_id": "uuid", + "claim_text": "string", + "support_status": "SUPPORTED | UNSUPPORTED | NEEDS_REVIEW", + "evidence_item_ids": [], + "risk_level": "low | medium | high" +} +``` + +## 7.8 ArticleDraft + +```json +{ + "id": "uuid", + "article_id": "uuid", + "version": 1, + "title": "string", + "slug": "string", + "meta_title": "string", + "meta_description": "string", + "body_markdown": "string", + "faq_block": [], + "schema_json": {}, + "internal_links": [], + "external_links": [], + "status": "DRAFT_ASSEMBLED" +} +``` + +## 7.9 Asset + +```json +{ + "id": "uuid", + "article_id": "uuid", + "asset_type": "hero_image | diagram | table | inline_image", + "title": "string", + "prompt": "string", + "file_url": "string", + "alt_text": "string", + "caption": "string", + "status": "PENDING | GENERATED | APPROVED | REJECTED" +} +``` + +## 7.10 WorkflowEvent + +```json +{ + "id": "uuid", + "article_id": "uuid", + "event_type": "PLAN_APPROVED", + "actor_type": "user | system | agent", + "actor_id": "uuid", + "payload": {}, + "created_at": "datetime" +} +``` + +## 7.11 AgentJob + +```json +{ + "id": "uuid", + "article_id": "uuid", + "job_type": "PLAN_GENERATION", + "agent_profile": "codex_subscription_default", + "status": "QUEUED | RUNNING | SUCCEEDED | FAILED | CANCELLED", + "workspace_path": "string", + "input_files": [], + "output_files": [], + "error_message": "string", + "started_at": "datetime", + "finished_at": "datetime" +} +``` + +## 7.12 ResearchRunManifest + +```json +{ + "id": "uuid", + "article_id": "uuid", + "agent_job_id": "uuid", + "s3_prefix": "research-runs/article_123/run_001/", + "artifacts": [ + { + "artifact_type": "source_section_snapshot", + "source_url": "https://example.com/article", + "object_key": "research-runs/article_123/run_001/source_001.json", + "content_hash": "sha256", + "metadata": { + "search_path": [], + "agent_selection_criteria": "string", + "head_metadata": {}, + "server_ip_address": "string", + "domain_whois_owner": "string" + } + } + ], + "created_at": "datetime" +} +``` + +Research source sections and metadata are stored indefinitely in object storage, not as large database blobs. Postgres stores manifests, object keys, hashes, source URLs, and artifact types. + +## 7.13 PublishCommit + +```json +{ + "id": "uuid", + "article_id": "uuid", + "target_site_id": "uuid", + "repository_url": "string", + "branch": "main", + "commit_sha": "string", + "content_bundle_manifest": {}, + "status": "PUBLISH_COMMIT_CREATED | PUBLISH_VERIFICATION_FAILED", + "deployment_status": "UNKNOWN | DISCOVERED | FAILED | SUCCEEDED", + "created_at": "datetime" +} +``` + +## 7.14 ScriptConfigVersion + +```json +{ + "id": "uuid", + "target_site_id": "uuid", + "version": 1, + "author_id": "uuid", + "yaml_config": {}, + "transform_script": "string", + "diff_from_previous": "string", + "rollback_target_version_id": "uuid", + "created_at": "datetime", + "activated_at": "datetime" +} +``` + +Every Admin script/config change stores author, timestamp, diff, and rollback target. v1 does not require second Admin approval for activation. + +--- + +# 8. Agent Runner Service + +## 8.1 Purpose + +Run subscription-authenticated local agents in isolated workspaces without exposing direct model API keys to the application. + +Primary runner: + +```text +Codex CLI with one managed ChatGPT subscription or enterprise access-token identity per environment +``` + +Optional runner: + +```text +Claude Code for internal-only workflows +``` + +## 8.2 Runner requirements + +```text +1. Create isolated workspace per job. +2. Write input files into workspace. +3. Run allowed CLI command. +4. Capture stdout, stderr, exit code. +5. Enforce timeout. +6. Validate output JSON against schema. +7. Store generated artifacts. +8. Return result to backend. +9. Record full job log. +10. Prevent access to unrelated article workspaces. +``` + +## 8.3 Example workspace + +```text +/workspaces/article_123/job_plan_generation_001/ + input/ + brief.json + target_site.json + boundary_answers.json + output_schema.json + instructions.md + output/ + plan.json + logs/ + stdout.log + stderr.log + command.json +``` + +## 8.4 Example Codex job command + +```bash +codex exec "Read input/instructions.md. Use input files. Produce output/plan.json only. Follow input/output_schema.json exactly." +``` + +## 8.5 Agent output validation + +All agent outputs must pass validation before workflow state changes. + +Validation rules: + +```text +1. JSON must be valid. +2. JSON must match expected schema. +3. Required fields must be present. +4. No empty section plan. +5. Every generated claim must be tied to a plan section. +6. Research output must include source URL, retrieval timestamp, artifact manifest reference, and content hash. +7. Draft cannot include unsupported factual claims unless marked for editor review. +8. Final approval cannot proceed while high-risk unsupported claims remain unresolved. +``` + +--- + +# 9. Workflow nodes + +## 9.1 Intake node + +Input: + +```json +{ + "brief_description": "string", + "target_site_id": "uuid", + "content_type": "longform_guide", + "primary_keyword": "optional string" +} +``` + +Output: + +```text +Article created +Initial workflow event stored +Status: ARTICLE_BRIEF_CREATED +``` + +Acceptance criteria: + +```text +Given a valid brief +When the user submits it +Then an article record is created +And workflow status is ARTICLE_BRIEF_CREATED +And target site config is attached +``` + +--- + +## 9.2 Boundary question node + +Purpose: + +Generate 5-10 questions to define scope, audience, angle, constraints, SEO, and evidence expectations. + +Required question categories: + +```text +Audience +Purpose +Reader outcome +Depth +Tone +Excluded topics +Primary keyword +Competitor angle +Evidence standard +Visual expectations +``` + +Output: + +```json +{ + "questions": [ + { + "question": "Who is the article for?", + "question_type": "audience", + "required": true + } + ] +} +``` + +Acceptance criteria: + +```text +Questions are generated from brief plus target site config. +Questions are editable before submission. +Required questions block plan generation until answered. +``` + +--- + +## 9.3 Plan generation node + +Purpose: + +Generate article production contract. + +Plan must include: + +```text +Title options +Recommended title +Reader persona +Search intent +Thesis +Detailed section outline +Claims to prove +Evidence needs +Visual needs +SEO notes +Tone guidance +Risks +Suggested internal links +``` + +Output schema: + +```json +{ + "title_options": ["string"], + "recommended_title": "string", + "reader_persona": "string", + "search_intent": "informational | commercial | navigational | transactional", + "thesis": "string", + "sections": [ + { + "heading": "string", + "purpose": "string", + "key_points": ["string"], + "claims_to_support": ["string"], + "evidence_needed": ["string"], + "visuals_needed": ["string"], + "target_word_count": 500 + } + ], + "seo": { + "primary_keyword": "string", + "secondary_keywords": ["string"], + "meta_title_draft": "string", + "meta_description_draft": "string" + }, + "risks": ["string"] +} +``` + +Acceptance criteria: + +```text +Plan contains at least 4 sections. +Each section has purpose, key points, and evidence needs. +Plan is saved as version 1. +Workflow stops at PLAN_REVIEW_REQUIRED. +No research starts before approval. +``` + +--- + +## 9.4 Plan review gate + +Purpose: + +Human approval checkpoint. + +Allowed editor actions: + +```text +Approve +Request revision +Edit directly +Reject article +Add notes +Change target site +Change content type +``` + +Acceptance criteria: + +```text +Workflow cannot continue until plan is approved. +Every approval or revision request is stored as WorkflowEvent. +Plan revisions create new versions. +Previous versions remain accessible. +``` + +--- + +## 9.5 Research node + +Purpose: + +Gather information from the web after plan approval through an explicit web search/fetch script, not free-form agent browsing alone. + +Inputs: + +```text +Approved plan +Target site source rules +Claims to prove +Required source types +Excluded domains +Editor notes +``` + +Output: + +```text +Research pack +Evidence matrix +Source summaries +Claim-source mapping +Unsupported claims list +Research run manifest with S3 object keys and content hashes +``` + +Research artifact storage: + +```text +1. Store broader discovered source content sections, not only final-used evidence snippets. +2. Store page metadata, search path to the article, agent criteria used to select it as evidence, page metadata, server IP address, and domain WHOIS owner. +3. Store source sections and metadata indefinitely in object storage. +4. Store only manifest records, S3 object keys, source URLs, content hashes, and artifact types in Postgres. +``` + +Evidence quality scoring: + +```text +Official documentation: high +Peer-reviewed research: high +Government or standards body: high +Reputable news/reporting: medium-high +Company blog: medium +Generic SEO blog: low-medium +Forum/reddit: context only +Unattributed content: low +``` + +Acceptance criteria: + +```text +Each factual claim has at least one source or is marked unsupported. +Each source has URL, title, domain, source type, summary, and retrieval timestamp. +Each research run writes an object-storage manifest. +If enough acceptable evidence cannot be found for the approved plan, workflow returns to PLAN_REVISION_REQUIRED before drafting. +Unsupported claims are visible in the editor UI. +``` + +--- + +## 9.6 Parallel production node + +Purpose: + +Run production tasks in parallel after evidence matrix is ready. + +Parallel tasks: + +```text +Section scaffold generation +Hero image prompt +Diagram specifications +Table specifications +FAQ block +SEO metadata +Internal link suggestions +Tone guide +``` + +Section scaffold output: + +```json +{ + "section_id": "uuid", + "heading": "string", + "draft_markdown": "string", + "used_evidence_ids": ["uuid"], + "unsupported_claims": ["string"], + "suggested_visuals": ["string"] +} +``` + +Acceptance criteria: + +```text +Every section from approved plan has one scaffold. +Each section lists used evidence IDs. +No scaffold silently introduces unsupported factual claims. +High-risk unsupported claims block final approval until resolved. +Parallel jobs can fail independently and be retried. +``` + +--- + +## 9.7 Visual asset node + +Purpose: + +Create visual production instructions and optionally generate assets. + +v1 requirement: + +```text +Generate prompts and specifications. +Store generated files if image tool is configured. +Allow editor approval or replacement. +``` + +Asset types: + +```text +Hero image +Inline diagram +Table +Flowchart +Comparison matrix +Architecture diagram +``` + +Diagram output example: + +```json +{ + "asset_type": "diagram", + "format": "mermaid", + "title": "AI content pipeline workflow", + "diagram_code": "graph TD; A[Brief] --> B[Questions];", + "alt_text": "Workflow showing article brief moving through planning, research, drafting, review, and publishing." +} +``` + +Acceptance criteria: + +```text +Each visual has title, type, prompt or code, alt text, and status. +Hero image prompt follows target site visual rules. +Tables and diagrams are linked to article sections. +``` + +--- + +## 9.8 Draft assembly node + +Purpose: + +Merge scaffolds, evidence, SEO metadata, and visual placeholders into one article draft. + +Markdown is the canonical editable draft format in v1. Rich preview renders Markdown/MDX for editor UX, but the stored approval artifact remains versioned Markdown plus metadata. + +Output: + +```text +Markdown draft +Metadata +Asset references +Evidence references +Unsupported claim warnings +``` + +Acceptance criteria: + +```text +Draft includes all planned sections. +Draft includes title, meta title, meta description, body, FAQ block if applicable, and visual placeholders. +Draft is saved as version 1. +``` + +--- + +## 9.9 SEO review node + +Purpose: + +Apply target-site SEO rules. + +Checks: + +```text +Title length +Meta title length +Meta description length +H1/H2 structure +Keyword placement +Internal links +External citations +Schema type +Slug +FAQ eligibility +Image alt text +Readability +Duplicate headings +``` + +SEO output: + +```json +{ + "score": 82, + "issues": [ + { + "severity": "medium", + "field": "meta_description", + "message": "Meta description is too long.", + "suggested_fix": "Shorten to 150-155 characters." + } + ], + "recommended_title": "string", + "recommended_slug": "string", + "schema_json": {} +} +``` + +Acceptance criteria: + +```text +SEO report is visible in final review. +Editor can accept or reject each SEO suggestion. +Target-site SEO rules override global rules. +``` + +--- + +## 9.10 Linguistic and tone review node + +Purpose: + +Check clarity, grammar, tone, brand voice, and consistency. + +Checks: + +```text +Grammar +Spelling +Sentence length +Passive voice +Jargon density +Brand tone +Forbidden phrases +Repetition +Weak claims +Unsupported certainty +CTA consistency +``` + +Acceptance criteria: + +```text +Issues are shown with severity and suggested rewrite. +Editor can accept, reject, or edit suggestions. +Final draft version stores accepted changes. +``` + +--- + +## 9.11 Final approval gate + +Purpose: + +Stop before Git-backed publish commit creation. + +Final approval checklist: + +```text +Plan followed +Evidence reviewed +Unsupported claims resolved +SEO metadata approved +Images approved +Tables and diagrams approved +Internal links approved +Frontmatter fields selected +Content path selected +Author selected +Publishing mode selected +``` + +Acceptance criteria: + +```text +Publish commit cannot be created until final approval is complete. +High-risk unsupported claims block final approval. +Approval event stores actor, timestamp, article version, and selected publishing settings. +``` + +--- + +## 9.12 Git-backed publish commit node + +Purpose: + +Create a content bundle and commit it directly to the configured production branch of a target Next.js site repository. + +Adapter interface: + +```python +class GitSitePublisher: + def build_content_bundle(self, article: ArticleDraft, config: dict) -> ContentBundle: + pass + + def run_transform(self, bundle: ContentBundle, site_workspace: str) -> TransformedBundle: + pass + + def dry_run_preview(self, bundle: TransformedBundle) -> DryRunResult: + pass + + def commit_to_production_branch(self, bundle: TransformedBundle) -> PublishCommit: + pass +``` + +v1 publishing target: + +```text +Git-backed Next.js content repository +Content bundle: Markdown/MDX file plus frontmatter JSON/YAML and referenced asset files +Primary done state: PUBLISH_COMMIT_CREATED +``` + +Acceptance criteria: + +```text +System materializes versioned site publishing YAML/scripts into the checked-out site repository workspace. +Admin-defined transformation scripts run directly on the runner host inside the checked-out site repository. +Final editor approval is required before publishing. +Best-effort dry-run validation uses the pipeline app's generic Markdown/MDX preview renderer, not the target site's actual build. +Accepted v1 validation risk: site-specific build/runtime errors may reach production. +System commits directly to the configured production branch. +Target repository's existing CI/CD handles deployment after commit. +Git conflicts or non-fast-forward pushes fail the publish step and require retry after refreshing from remote. +If delayed deployment verification fails, system alerts Admin/Editor and leaves the production commit in place. +System stores commit SHA, repository URL, branch, content bundle manifest, and publish status. +``` + +--- + +# 10. Multi-site configuration + +## 10.1 Site config structure + +```json +{ + "site_id": "b2b_saas_blog", + "name": "B2B SaaS Blog", + "publishing_target": { + "type": "git_next", + "repository_url": "git@github.com:example/site.git", + "production_branch": "main", + "commit_author_name": "AI Content Pipeline", + "commit_author_email": "pipeline@example.com" + }, + "editorial": { + "audience": "B2B SaaS founders and content leads", + "tone": "direct, expert, practical", + "forbidden_phrases": ["revolutionary", "game-changing"], + "preferred_structure": ["intro", "framework", "examples", "implementation", "conclusion"] + }, + "seo": { + "title_max_chars": 60, + "meta_description_max_chars": 155, + "slug_style": "kebab-case", + "schema_type": "Article" + }, + "sources": { + "preferred_domains": ["official docs", "research papers", "government sources"], + "excluded_domains": [], + "minimum_sources": 5 + }, + "visuals": { + "hero_ratio": "16:9", + "style": "clean editorial illustration", + "diagram_format": "mermaid" + }, + "publishing": { + "content_format": "mdx", + "content_path_template": "content/articles/{slug}.mdx", + "asset_path_template": "public/articles/{slug}/{filename}", + "frontmatter_mapping": { + "title": "title", + "description": "description", + "date": "date", + "author": "author", + "category": "category", + "tags": "tags" + }, + "transform_script_version_id": "uuid", + "dry_run_renderer": "generic_mdx_preview" + } +} +``` + +## 10.2 Acceptance criteria + +```text +Adding a new website must not require workflow code changes. +Each site can define repository target, tone, SEO, visual, source, publishing, frontmatter, asset-path, and transformation rules. +Workflow reads target site config at every generation step. +``` + +--- + +# 11. API endpoints + +## Articles + +```http +POST /api/articles +GET /api/articles +GET /api/articles/{article_id} +PATCH /api/articles/{article_id} +DELETE /api/articles/{article_id} +``` + +## Boundary questions + +```http +POST /api/articles/{article_id}/boundary-questions/generate +GET /api/articles/{article_id}/boundary-questions +PATCH /api/articles/{article_id}/boundary-questions/{question_id} +POST /api/articles/{article_id}/boundary-questions/submit +``` + +## Plans + +```http +POST /api/articles/{article_id}/plan/generate +GET /api/articles/{article_id}/plans +GET /api/articles/{article_id}/plans/{plan_id} +POST /api/articles/{article_id}/plans/{plan_id}/approve +POST /api/articles/{article_id}/plans/{plan_id}/request-revision +PATCH /api/articles/{article_id}/plans/{plan_id} +``` + +## Research + +```http +POST /api/articles/{article_id}/research/start +GET /api/articles/{article_id}/research +GET /api/articles/{article_id}/evidence +PATCH /api/articles/{article_id}/evidence/{evidence_id} +``` + +## Drafts + +```http +POST /api/articles/{article_id}/draft/assemble +GET /api/articles/{article_id}/drafts +GET /api/articles/{article_id}/drafts/{draft_id} +PATCH /api/articles/{article_id}/drafts/{draft_id} +``` + +## SEO and language + +```http +POST /api/articles/{article_id}/seo/review +GET /api/articles/{article_id}/seo/report +POST /api/articles/{article_id}/language/review +GET /api/articles/{article_id}/language/report +``` + +## Assets + +```http +POST /api/articles/{article_id}/assets/generate-specs +GET /api/articles/{article_id}/assets +PATCH /api/articles/{article_id}/assets/{asset_id} +POST /api/articles/{article_id}/assets/{asset_id}/approve +``` + +## Final approval + +```http +POST /api/articles/{article_id}/final-approval +POST /api/articles/{article_id}/final-revision-request +``` + +## Publishing + +```http +POST /api/articles/{article_id}/publishing/dry-run +POST /api/articles/{article_id}/publishing/create-commit +GET /api/articles/{article_id}/publishing/status +GET /api/articles/{article_id}/publishing/commits +``` + +## Site config + +```http +POST /api/sites +GET /api/sites +GET /api/sites/{site_id} +PATCH /api/sites/{site_id} +DELETE /api/sites/{site_id} +GET /api/sites/{site_id}/publishing-config/versions +POST /api/sites/{site_id}/publishing-config/versions +POST /api/sites/{site_id}/publishing-config/versions/{version_id}/activate +POST /api/sites/{site_id}/publishing-config/versions/{version_id}/rollback +``` + +## Agent jobs + +```http +GET /api/agent-jobs +GET /api/agent-jobs/{job_id} +POST /api/agent-jobs/{job_id}/retry +POST /api/agent-jobs/{job_id}/cancel +``` + +--- + +# 12. Prompt and instruction management + +## Prompt modules + +```text +boundary_questions.md +plan_generation.md +research_brief.md +evidence_matrix.md +section_scaffold.md +hero_image_prompt.md +table_spec.md +diagram_spec.md +seo_review.md +language_review.md +draft_assembly.md +publish_bundle.md +``` + +## Prompt versioning requirements + +```text +Each prompt has version. +Each agent job stores prompt version. +Article history shows which prompt version generated each artifact. +Admin can activate/deactivate prompt versions. +``` + +## Prompt input format + +All prompts receive structured input: + +```json +{ + "article": {}, + "target_site": {}, + "boundary_answers": {}, + "approved_plan": {}, + "editor_notes": [], + "evidence": [], + "output_schema": {} +} +``` + +--- + +# 13. Security requirements + +## Credentials + +```text +1. Do not store OpenAI or Anthropic API keys for generation in the app. +2. One managed Codex CLI subscription or enterprise access-token identity lives only on the runner host per environment. +3. Git credentials for target repositories are stored in a secrets manager or runner-local secure credential store. +4. The backend stores references to credentials, not raw secrets. +5. Agent generation jobs run in isolated workspaces with limited filesystem access. +6. Admin-defined publishing transform scripts are trusted host-level code and may run directly in checked-out site repositories. +``` + +## Workspace isolation + +```text +1. One workspace per agent job. +2. Workspace path must be generated by backend. +3. Runner cannot access sibling workspaces. +4. Job artifacts are copied to object storage after completion. +5. Workspace can be deleted after retention period. +``` + +## Publishing safety + +```text +1. v1 commits directly to the configured production branch. +2. Final approval is required before publish commit creation. +3. A generic Markdown/MDX content-shape dry run must pass before commit. +4. The dry run is best-effort and does not guarantee target-site build/runtime correctness. +5. Git conflicts and non-fast-forward pushes fail instead of auto-rebasing. +6. All publishing actions and script/config versions are logged. +``` + +--- + +# 14. Error handling + +## Agent job errors + +```text +FAILED_SCHEMA_VALIDATION +CLI_EXIT_CODE_FAILURE +TIMEOUT +MISSING_OUTPUT_FILE +UNSUPPORTED_CLAIMS_FOUND +SOURCE_RETRIEVAL_FAILED +RESEARCH_ARTIFACT_UPLOAD_FAILED +PUBLISH_DRY_RUN_FAILED +GIT_CHECKOUT_FAILED +GIT_COMMIT_FAILED +GIT_PUSH_NON_FAST_FORWARD +PUBLISH_VERIFICATION_FAILED +``` + +## Retry rules + +```text +Plan generation: retry manually only +Research: retry allowed +Section scaffold: retry allowed per section +SEO review: retry allowed +Language review: retry allowed +Publish commit creation: retry allowed only if no publish commit exists +``` + +## Failure UI + +Each failure should show: + +```text +Job type +Status +Error category +Error message +Last successful step +Retry button +Logs for admin +``` + +--- + +# 15. Audit trail + +Every important action must be logged: + +```text +Article created +Boundary questions generated +Boundary answers submitted +Plan generated +Plan edited +Plan approved +Research started +Evidence added +Draft assembled +SEO review completed +Language review completed +Asset approved +Final approval granted +Publishing dry run completed +Publish commit created +Delayed publish verification failed +Site publishing config changed +Site transform script changed +Job failed +Job retried +``` + +Audit event fields: + +```json +{ + "event_type": "PLAN_APPROVED", + "actor": "user_id", + "article_id": "uuid", + "payload": {}, + "created_at": "datetime" +} +``` + +--- + +# 16. Development tasks + +## Epic 1: Project foundation + +### Task 1.1: Initialize repositories + +Deliverables: + +```text +Frontend app +Backend app +Runner service +Shared schemas package +Docker compose environment +``` + +Acceptance criteria: + +```text +Developer can run full local stack with one command. +Frontend can call backend health endpoint. +Backend can connect to Postgres. +Runner service can receive test job. +``` + +### Task 1.2: Database schema + +Deliverables: + +```text +Postgres migrations +Core tables +Indexes +Seed site config +``` + +Tables: + +```text +users +target_sites +articles +boundary_questions +article_plans +plan_sections +evidence_items +claims +article_drafts +assets +workflow_events +agent_jobs +research_run_manifests +publish_commits +script_config_versions +prompt_versions +``` + +Acceptance criteria: + +```text +Migrations run cleanly. +Seed creates one sample target site. +Article can be created and queried. +``` + +--- + +## Epic 2: Article intake and boundary questions + +### Task 2.1: Article creation API + +Acceptance criteria: + +```text +POST /api/articles creates article. +Article status is ARTICLE_BRIEF_CREATED. +Target site config is attached. +WorkflowEvent is created. +``` + +### Task 2.2: Boundary question generation + +Acceptance criteria: + +```text +System generates questions based on brief and site config. +Questions are stored. +User can edit answers. +Required unanswered questions block plan generation. +``` + +### Task 2.3: Boundary questions UI + +Acceptance criteria: + +```text +User can submit brief. +User can answer generated questions. +User can save partial answers. +User can submit answers and continue. +``` + +--- + +## Epic 3: Plan generation and approval + +### Task 3.1: Codex runner MVP + +Acceptance criteria: + +```text +Backend can create AgentJob. +Runner creates isolated workspace. +Runner writes input files. +Runner invokes Codex CLI. +Runner captures logs. +Runner validates output JSON. +Runner updates job status. +``` + +### Task 3.2: Plan generation workflow node + +Acceptance criteria: + +```text +Plan generated from brief, boundary answers, and site config. +Plan saved as version 1. +Workflow stops at PLAN_REVIEW_REQUIRED. +``` + +### Task 3.3: Plan review UI + +Acceptance criteria: + +```text +Editor can review full plan. +Editor can edit plan. +Editor can request revision. +Editor can approve plan. +Approval creates WorkflowEvent. +``` + +--- + +## Epic 4: Research and evidence + +### Task 4.1: Research job + +Acceptance criteria: + +```text +Research runs only after approved plan. +Research gathers sources for claims through explicit search/fetch scripts. +Evidence items are stored. +Each evidence item includes URL, title, source type, summary, quality score, retrieval timestamp. +Research source sections and expanded metadata are uploaded to object storage. +Research run manifest stores S3 keys, content hashes, source URLs, and artifact types. +Insufficient acceptable evidence forces PLAN_REVISION_REQUIRED before drafting. +``` + +### Task 4.2: Evidence matrix + +Acceptance criteria: + +```text +System maps claims to evidence. +Unsupported claims are marked. +Editor can view evidence by section. +Editor can add or remove evidence manually. +``` + +### Task 4.3: Evidence UI + +Acceptance criteria: + +```text +Editor can filter by section. +Editor can filter unsupported claims. +Editor can open source URLs. +Editor can mark evidence as approved or rejected. +``` + +--- + +## Epic 5: Parallel production + +### Task 5.1: Section scaffold jobs + +Acceptance criteria: + +```text +One job is created per approved plan section. +Jobs run independently. +Each output includes draft markdown, evidence IDs, unsupported claims, and suggested visuals. +Failed section can be retried alone. +``` + +### Task 5.2: Asset spec jobs + +Acceptance criteria: + +```text +System creates hero image prompt. +System creates table specs. +System creates diagram specs. +Each asset is linked to section or article. +``` + +### Task 5.3: Draft assembly + +Acceptance criteria: + +```text +All successful section scaffolds are merged. +Draft includes metadata and visual placeholders. +Draft is saved as version 1. +Workflow moves to SEO_AND_LANGUAGE_REVIEW_READY. +``` + +--- + +## Epic 6: SEO and linguistic review + +### Task 6.1: SEO review job + +Acceptance criteria: + +```text +SEO report generated from target site rules. +Report includes score, issues, suggested fixes, metadata, slug, schema. +Editor can accept or reject suggestions. +``` + +### Task 6.2: Language review job + +Acceptance criteria: + +```text +Language report includes grammar, clarity, tone, repetition, and forbidden phrase issues. +Suggestions are linked to exact draft locations. +Editor can accept or reject suggestions. +``` + +### Task 6.3: Final review UI + +Acceptance criteria: + +```text +Editor sees draft, assets, SEO report, language report, and evidence warnings. +Editor can edit final draft. +Editor can approve final version. +Workflow stops until final approval. +``` + +--- + +## Epic 7: Git-backed Next publishing + +### Task 7.1: Git publisher interface + +Acceptance criteria: + +```text +GitSitePublisher interface implemented. +Publisher builds Markdown/MDX content bundles with frontmatter and referenced assets. +Publisher materializes versioned YAML config and transform scripts into the site repository workspace. +Publisher captures transform logs and output manifest. +``` + +### Task 7.2: Publish commit creation + +Acceptance criteria: + +```text +Publish commit creation requires final approval. +Generic Markdown/MDX dry-run preview must pass. +System commits directly to the configured production branch. +Non-fast-forward push or conflict fails the publish step. +Commit SHA, repository URL, branch, and content bundle manifest are stored. +Publishing event is logged. +``` + +### Task 7.3: Publishing UI + +Acceptance criteria: + +```text +Editor can edit frontmatter, content path, author, category, and tags. +Editor can run generic Markdown/MDX preview. +Editor can click Create publish commit. +Publishing result shows PUBLISH_COMMIT_CREATED, commit SHA, and opportunistic deployment status if discoverable. +``` + +--- + +## Epic 8: Multi-site support + +### Task 8.1: Site config CRUD + +Acceptance criteria: + +```text +Admin can create, edit, and deactivate target sites. +Site config includes editorial, SEO, source, visual, publishing, Git repository, frontmatter mapping, asset path, and transform script rules. +Every Admin config/script change stores author, timestamp, diff, and rollback target. +``` + +### Task 8.2: Site-aware workflow + +Acceptance criteria: + +```text +Every generation step receives target site config. +Different sites produce different tone, SEO metadata, visual specs, frontmatter, asset paths, and content bundles. +No hardcoded website logic exists outside adapters. +``` + +--- + +## Epic 9: Observability and operations + +### Task 9.1: Job logs + +Acceptance criteria: + +```text +Each agent job stores stdout, stderr, status, duration, and error category. +Admin can inspect logs. +Sensitive values are redacted. +``` + +### Task 9.2: Workflow history + +Acceptance criteria: + +```text +Article detail page shows workflow timeline. +Timeline includes user actions, system actions, and agent jobs. +``` + +### Task 9.3: Retry and cancel + +Acceptance criteria: + +```text +Admin/editor can retry failed jobs where allowed. +Admin/editor can cancel queued or running jobs. +Cancelled jobs do not update article state. +``` + +--- + +# 17. MVP milestone plan + +## Milestone 1: Foundation + +Goal: + +```text +Local stack, database, article creation, target site config. +``` + +Exit criteria: + +```text +User can create article brief for one target website. +Article appears in dashboard. +``` + +## Milestone 2: Plan approval loop + +Goal: + +```text +Boundary questions, Codex runner, plan generation, plan review. +``` + +Exit criteria: + +```text +Editor can approve or revise an article plan. +Workflow stops correctly before research. +``` + +## Milestone 3: Research and evidence + +Goal: + +```text +Research job, evidence matrix, claim-source mapping, object-storage research artifacts. +``` + +Exit criteria: + +```text +Approved plan produces evidence matrix. +Research run manifest points to S3 source-section artifacts. +Unsupported claims are visible. +Insufficient evidence returns workflow to plan revision. +``` + +## Milestone 4: Draft production + +Goal: + +```text +Parallel section scaffolding, asset specs, draft assembly. +``` + +Exit criteria: + +```text +System creates full markdown draft from approved plan and evidence. +``` + +## Milestone 5: Review and Git-backed publish commit + +Goal: + +```text +SEO review, linguistic review, final approval, generic preview dry run, Git-backed publish commit. +``` + +Exit criteria: + +```text +Editor can approve final article and create a production-branch publish commit. +``` + +## Milestone 6: Multi-site extension + +Goal: + +```text +Second target website with different config. +``` + +Exit criteria: + +```text +Same workflow works for two websites through site config and versioned publishing YAML/scripts. +``` + +--- + +# 18. Definition of done for v1 + +```text +1. User can create article from short description. +2. System generates boundary questions. +3. User can answer questions. +4. System generates article plan. +5. Workflow stops for plan approval. +6. Editor can revise or approve plan. +7. Research starts only after approval. +8. Evidence matrix is created. +9. Section scaffolds run in parallel. +10. Visual specs are generated. +11. Draft is assembled. +12. SEO review is generated. +13. Linguistic review is generated. +14. Workflow stops for final approval. +15. Publish commit is created only after final approval and successful generic dry-run validation. +16. At least one target website works end to end. +17. A second target website can be added through config. +18. Agent generation runs through Codex CLI subscription auth, not direct API-key calls. +19. Every major action is logged. +20. Failed jobs can be inspected and retried. +21. High-risk unsupported claims block final approval. +22. Research source-section artifacts are stored in S3 with manifest records in Postgres. +23. Publishing validation is explicitly best-effort content-shape validation only. +``` + +--- + +# 19. Technical risks and mitigations + +| Risk | Impact | Mitigation | +| ----------------------------------------------------------- | -----: | --------------------------------------------------------------------------- | +| Codex CLI output is malformed | High | Use schema files, JSON-only instructions, validation, retry | +| CLI auth expires | High | Runner health check, admin alert, re-auth flow | +| Long jobs block backend | High | Use queue and runner service, not request thread | +| Research creates weak evidence | High | Source scoring, unsupported claim flags, plan revision loop, editor review | +| Expanded research corpus creates storage/legal burden | Medium | Store artifacts in S3 with manifests and hashes; document indefinite retention | +| Multi-site rules become messy | Medium | Strict site config schema, versioned YAML/scripts, audit diffs | +| Direct production-branch commit contains site-specific error | High | Final approval, generic dry run, explicit accepted validation risk, alerts | +| Git push conflict or non-fast-forward | Medium | Fail publish step and require retry after refresh | +| Admin transform script damages runner/site workspace | High | Treat Admins as trusted code operators, audit every version, rollback | +| Agent invents unsupported claims | High | Claim extraction, evidence mapping, unsupported-claim gate | +| Claude Code subscription use creates compliance issue | Medium | Keep Claude optional and internal-only | +| Prompt changes break output | Medium | Prompt versioning and schema validation | + +--- + +# 20. First development ticket + +## Ticket: Build v1 foundation for AI Content Pipeline + +### Goal + +Create the initial backend, frontend, database, and runner-service foundation for the article workflow. + +### Scope + +Implement: + +```text +1. Article creation +2. Target site configuration with Git-backed publishing fields +3. Workflow state storage +4. Agent job table +5. Codex runner test job +6. Dashboard list +7. Article detail shell +``` + +### Backend deliverables + +```text +FastAPI app +Postgres migrations +Article model +TargetSite model +WorkflowEvent model +AgentJob model +ScriptConfigVersion model +POST /api/articles +GET /api/articles +GET /api/articles/{id} +GET /api/sites +POST /api/sites +POST /api/agent-jobs/test-codex +``` + +### Frontend deliverables + +```text +Dashboard page +New article form +Article detail page +Status badge component +Target site selector +Publishing status display +``` + +### Runner deliverables + +```text +Runner service process +Workspace creation +Input file writer +Codex CLI command execution +Log capture +Output validation placeholder +Job status update +``` + +### Acceptance criteria + +```text +Given a configured target site +When a user submits a short article brief +Then the system creates an article with ARTICLE_BRIEF_CREATED status +And the article appears on the dashboard +And the article detail page shows workflow history +And the target site can store repository, production branch, and initial publishing YAML config +And an admin can run a test Codex job from backend +And the runner stores stdout, stderr, exit code, and job status +``` + +--- + +# 21. Recommended implementation order + +```text +1. Database schema +2. Backend article/site APIs +3. Frontend dashboard and article form +4. Workflow event logging +5. Agent job model +6. Runner service +7. Codex CLI test job +8. Boundary question generation +9. Plan generation +10. Plan review gate +11. Research and evidence +12. Parallel scaffolding +13. Draft assembly +14. SEO and language review +15. Git-backed publishing adapter +16. Second target website config +``` diff --git a/Makefile b/Makefile new file mode 100644 index 0000000..1d67cda --- /dev/null +++ b/Makefile @@ -0,0 +1,10 @@ +.PHONY: dev down smoke + +dev: + docker compose up --build + +down: + docker compose down --remove-orphans + +smoke: + bash tests/smoke/public-health.sh diff --git a/README.md b/README.md new file mode 100644 index 0000000..866f869 --- /dev/null +++ b/README.md @@ -0,0 +1,41 @@ +# AI Content Pipeline + +Dockerized monorepo foundation for the AI Content Pipeline. + +## Local Development + +Start the full stack from the repository root: + +```sh +make dev +``` + +Equivalent command: + +```sh +docker compose up --build +``` + +The stack exposes: + +| Service | URL | +| --- | --- | +| Frontend | `http://localhost:3000` | +| Backend | `http://localhost:8000` | +| Runner | `http://localhost:8010` | +| MinIO console | `http://localhost:9001` | + +Local-only defaults are documented in `.env.example`. Copy it to `.env` only when +you need to override local ports or service credentials. + +## Health Smoke + +Run the public health smoke test: + +```sh +make smoke +``` + +The smoke test starts Docker Compose, checks frontend/backend/runner health +endpoints, and verifies that the backend can connect to Postgres, Redis, and +S3-compatible object storage. diff --git a/apps/backend/Dockerfile b/apps/backend/Dockerfile new file mode 100644 index 0000000..bf6a379 --- /dev/null +++ b/apps/backend/Dockerfile @@ -0,0 +1,15 @@ +FROM python:3.12-slim + +ENV PYTHONDONTWRITEBYTECODE=1 +ENV PYTHONUNBUFFERED=1 + +WORKDIR /app + +COPY apps/backend/requirements.txt ./requirements.txt +RUN pip install --no-cache-dir -r requirements.txt + +COPY apps/backend/src ./src + +EXPOSE 8000 + +CMD ["uvicorn", "src.presentation.main:app", "--host", "0.0.0.0", "--port", "8000"] diff --git a/apps/backend/requirements.txt b/apps/backend/requirements.txt new file mode 100644 index 0000000..c5282c4 --- /dev/null +++ b/apps/backend/requirements.txt @@ -0,0 +1,5 @@ +fastapi==0.115.6 +uvicorn[standard]==0.34.0 +psycopg[binary]==3.2.3 +redis==5.2.1 +boto3==1.35.90 diff --git a/apps/backend/src/__init__.py b/apps/backend/src/__init__.py new file mode 100644 index 0000000..a1b671c --- /dev/null +++ b/apps/backend/src/__init__.py @@ -0,0 +1 @@ +"""Backend service package.""" diff --git a/apps/backend/src/application/README.md b/apps/backend/src/application/README.md new file mode 100644 index 0000000..648c0a6 --- /dev/null +++ b/apps/backend/src/application/README.md @@ -0,0 +1,3 @@ +# Application + +Application use cases and workflow orchestration belong in this layer. diff --git a/apps/backend/src/application/__init__.py b/apps/backend/src/application/__init__.py new file mode 100644 index 0000000..ddf881d --- /dev/null +++ b/apps/backend/src/application/__init__.py @@ -0,0 +1 @@ +"""Application layer for use cases and orchestration.""" diff --git a/apps/backend/src/domain/README.md b/apps/backend/src/domain/README.md new file mode 100644 index 0000000..1573d79 --- /dev/null +++ b/apps/backend/src/domain/README.md @@ -0,0 +1,3 @@ +# Domain + +Business entities, value objects, and domain rules belong in this layer. diff --git a/apps/backend/src/domain/__init__.py b/apps/backend/src/domain/__init__.py new file mode 100644 index 0000000..6759e85 --- /dev/null +++ b/apps/backend/src/domain/__init__.py @@ -0,0 +1 @@ +"""Domain layer for business concepts and invariants.""" diff --git a/apps/backend/src/infrastructure/README.md b/apps/backend/src/infrastructure/README.md new file mode 100644 index 0000000..118813a --- /dev/null +++ b/apps/backend/src/infrastructure/README.md @@ -0,0 +1,4 @@ +# Infrastructure + +Database, cache, object storage, queue, and other external adapters belong in +this layer. diff --git a/apps/backend/src/infrastructure/__init__.py b/apps/backend/src/infrastructure/__init__.py new file mode 100644 index 0000000..5995f77 --- /dev/null +++ b/apps/backend/src/infrastructure/__init__.py @@ -0,0 +1 @@ +"""Infrastructure adapters for external services.""" diff --git a/apps/backend/src/infrastructure/dependencies.py b/apps/backend/src/infrastructure/dependencies.py new file mode 100644 index 0000000..34b0b23 --- /dev/null +++ b/apps/backend/src/infrastructure/dependencies.py @@ -0,0 +1,74 @@ +from __future__ import annotations + +import os +from collections.abc import Callable + +import boto3 +import psycopg +from botocore.config import Config +from redis import Redis + + +DependencyStatus = dict[str, str] + + +def check_dependencies() -> DependencyStatus: + checks: dict[str, Callable[[], None]] = { + "postgres": _check_postgres, + "redis": _check_redis, + "object_storage": _check_object_storage, + } + + results: DependencyStatus = {} + for name, check in checks.items(): + try: + check() + except Exception: + results[name] = "error" + else: + results[name] = "ok" + + return results + + +def _check_postgres() -> None: + dsn = _required_env("POSTGRES_DSN") + with psycopg.connect(dsn, connect_timeout=3) as connection: + with connection.cursor() as cursor: + cursor.execute("SELECT 1") + cursor.fetchone() + + +def _check_redis() -> None: + redis_url = _required_env("REDIS_URL") + client = Redis.from_url(redis_url, socket_connect_timeout=3, socket_timeout=3) + try: + if client.ping() is not True: + raise RuntimeError("Redis ping failed") + finally: + client.close() + + +def _check_object_storage() -> None: + endpoint = _required_env("OBJECT_STORAGE_ENDPOINT") + access_key_id = _required_env("OBJECT_STORAGE_ACCESS_KEY_ID") + secret_access_key = _required_env("OBJECT_STORAGE_SECRET_ACCESS_KEY") + bucket = _required_env("OBJECT_STORAGE_BUCKET") + region = os.environ.get("OBJECT_STORAGE_REGION", "us-east-1") + + client = boto3.client( + "s3", + endpoint_url=endpoint, + aws_access_key_id=access_key_id, + aws_secret_access_key=secret_access_key, + region_name=region, + config=Config(signature_version="s3v4", connect_timeout=3, read_timeout=3), + ) + client.head_bucket(Bucket=bucket) + + +def _required_env(name: str) -> str: + value = os.environ.get(name) + if not value: + raise RuntimeError(f"Missing required environment variable: {name}") + return value diff --git a/apps/backend/src/presentation/README.md b/apps/backend/src/presentation/README.md new file mode 100644 index 0000000..3bee2aa --- /dev/null +++ b/apps/backend/src/presentation/README.md @@ -0,0 +1,3 @@ +# Presentation + +HTTP routes, request/response DTOs, and API wiring belong in this layer. diff --git a/apps/backend/src/presentation/__init__.py b/apps/backend/src/presentation/__init__.py new file mode 100644 index 0000000..4f8115f --- /dev/null +++ b/apps/backend/src/presentation/__init__.py @@ -0,0 +1 @@ +"""Presentation layer for HTTP APIs.""" diff --git a/apps/backend/src/presentation/main.py b/apps/backend/src/presentation/main.py new file mode 100644 index 0000000..3127ac2 --- /dev/null +++ b/apps/backend/src/presentation/main.py @@ -0,0 +1,27 @@ +from fastapi import FastAPI +from fastapi.responses import JSONResponse + +from src.infrastructure.dependencies import check_dependencies + + +app = FastAPI(title="AI Content Pipeline Backend") + + +@app.get("/health") +def health() -> dict[str, str]: + return {"service": "backend", "status": "ok"} + + +@app.get("/health/dependencies") +def dependency_health() -> JSONResponse: + dependencies = check_dependencies() + is_ok = all(status == "ok" for status in dependencies.values()) + + return JSONResponse( + status_code=200 if is_ok else 503, + content={ + "service": "backend", + "status": "ok" if is_ok else "error", + "dependencies": dependencies, + }, + ) diff --git a/apps/backend/src/shared/README.md b/apps/backend/src/shared/README.md new file mode 100644 index 0000000..9a06623 --- /dev/null +++ b/apps/backend/src/shared/README.md @@ -0,0 +1,4 @@ +# Shared + +Cross-layer backend primitives belong here. Keep this layer small and avoid +placing domain rules in shared code. diff --git a/apps/backend/src/shared/__init__.py b/apps/backend/src/shared/__init__.py new file mode 100644 index 0000000..0945bcc --- /dev/null +++ b/apps/backend/src/shared/__init__.py @@ -0,0 +1 @@ +"""Shared backend primitives.""" diff --git a/apps/frontend/ARCHITECTURE.md b/apps/frontend/ARCHITECTURE.md new file mode 100644 index 0000000..e2bacf5 --- /dev/null +++ b/apps/frontend/ARCHITECTURE.md @@ -0,0 +1,17 @@ +# Frontend Architecture + +This app uses Next.js App Router in `src/app/` for routing. Feature-Sliced +Design layers live next to it under `src/`: + +```text +src/app/ +src/processes/ +src/pages/ +src/widgets/ +src/features/ +src/entities/ +src/shared/ +``` + +`src/pages` is an FSD page-composition layer. It is not the legacy Next.js Pages +Router; the App Router is `src/app`. diff --git a/apps/frontend/Dockerfile b/apps/frontend/Dockerfile new file mode 100644 index 0000000..b693e1a --- /dev/null +++ b/apps/frontend/Dockerfile @@ -0,0 +1,12 @@ +FROM node:22-alpine + +WORKDIR /app + +COPY apps/frontend/package.json ./package.json +RUN npm install + +COPY apps/frontend ./ + +EXPOSE 3000 + +CMD ["npm", "run", "dev", "--", "--hostname", "0.0.0.0"] diff --git a/apps/frontend/next-env.d.ts b/apps/frontend/next-env.d.ts new file mode 100644 index 0000000..206106a --- /dev/null +++ b/apps/frontend/next-env.d.ts @@ -0,0 +1,5 @@ +/// +/// + +// This file is generated by Next.js and kept here so TypeScript works before +// the first local `next dev` run. diff --git a/apps/frontend/next.config.mjs b/apps/frontend/next.config.mjs new file mode 100644 index 0000000..d5456a1 --- /dev/null +++ b/apps/frontend/next.config.mjs @@ -0,0 +1,6 @@ +/** @type {import('next').NextConfig} */ +const nextConfig = { + reactStrictMode: true, +}; + +export default nextConfig; diff --git a/apps/frontend/package.json b/apps/frontend/package.json new file mode 100644 index 0000000..89c0ca3 --- /dev/null +++ b/apps/frontend/package.json @@ -0,0 +1,23 @@ +{ + "name": "@pipeline/frontend", + "private": true, + "scripts": { + "dev": "next dev", + "build": "next build", + "start": "next start" + }, + "dependencies": { + "next": "16.2.6", + "react": "19.0.0", + "react-dom": "19.0.0" + }, + "devDependencies": { + "@types/node": "22.10.5", + "@types/react": "19.0.2", + "@types/react-dom": "19.0.2", + "typescript": "5.7.2" + }, + "overrides": { + "postcss": "8.5.14" + } +} diff --git a/apps/frontend/src/app/globals.css b/apps/frontend/src/app/globals.css new file mode 100644 index 0000000..348c869 --- /dev/null +++ b/apps/frontend/src/app/globals.css @@ -0,0 +1,21 @@ +:root { + color-scheme: light; + font-family: + Inter, ui-sans-serif, system-ui, -apple-system, BlinkMacSystemFont, "Segoe UI", + sans-serif; + background: #f6f7f9; + color: #1d2733; +} + +* { + box-sizing: border-box; +} + +body { + margin: 0; +} + +main { + min-height: 100vh; + padding: 48px; +} diff --git a/apps/frontend/src/app/health/route.ts b/apps/frontend/src/app/health/route.ts new file mode 100644 index 0000000..339d295 --- /dev/null +++ b/apps/frontend/src/app/health/route.ts @@ -0,0 +1,5 @@ +export const dynamic = "force-dynamic"; + +export async function GET() { + return Response.json({ service: "frontend", status: "ok" }); +} diff --git a/apps/frontend/src/app/layout.tsx b/apps/frontend/src/app/layout.tsx new file mode 100644 index 0000000..ea66226 --- /dev/null +++ b/apps/frontend/src/app/layout.tsx @@ -0,0 +1,19 @@ +import "./globals.css"; +import type { ReactNode } from "react"; + +export const metadata = { + title: "AI Content Pipeline", + description: "Internal editorial workflow foundation", +}; + +export default function RootLayout({ + children, +}: Readonly<{ + children: ReactNode; +}>) { + return ( + + {children} + + ); +} diff --git a/apps/frontend/src/app/page.tsx b/apps/frontend/src/app/page.tsx new file mode 100644 index 0000000..3b5dfd7 --- /dev/null +++ b/apps/frontend/src/app/page.tsx @@ -0,0 +1,8 @@ +export default function HomePage() { + return ( +
+

AI Content Pipeline

+

Dockerized monorepo foundation is running.

+
+ ); +} diff --git a/apps/frontend/src/entities/README.md b/apps/frontend/src/entities/README.md new file mode 100644 index 0000000..5f293a3 --- /dev/null +++ b/apps/frontend/src/entities/README.md @@ -0,0 +1,3 @@ +# Entities + +FSD layer for domain entity UI and client-side models. diff --git a/apps/frontend/src/features/README.md b/apps/frontend/src/features/README.md new file mode 100644 index 0000000..1c6ec33 --- /dev/null +++ b/apps/frontend/src/features/README.md @@ -0,0 +1,3 @@ +# Features + +FSD layer for user-facing actions and feature logic. diff --git a/apps/frontend/src/pages/README.md b/apps/frontend/src/pages/README.md new file mode 100644 index 0000000..86a8c1d --- /dev/null +++ b/apps/frontend/src/pages/README.md @@ -0,0 +1,4 @@ +# Pages + +FSD page-composition layer. This is separate from the Next.js App Router +`app/` route layer and does not enable the legacy Pages Router. diff --git a/apps/frontend/src/processes/README.md b/apps/frontend/src/processes/README.md new file mode 100644 index 0000000..02d3dbe --- /dev/null +++ b/apps/frontend/src/processes/README.md @@ -0,0 +1,4 @@ +# Processes + +FSD layer for cross-page user workflows. Placeholder for upcoming article +workflow processes. diff --git a/apps/frontend/src/shared/README.md b/apps/frontend/src/shared/README.md new file mode 100644 index 0000000..6bfe8f4 --- /dev/null +++ b/apps/frontend/src/shared/README.md @@ -0,0 +1,3 @@ +# Shared + +FSD layer for reusable frontend primitives, API clients, and UI foundations. diff --git a/apps/frontend/src/widgets/README.md b/apps/frontend/src/widgets/README.md new file mode 100644 index 0000000..2249e33 --- /dev/null +++ b/apps/frontend/src/widgets/README.md @@ -0,0 +1,3 @@ +# Widgets + +FSD layer for composed UI blocks used by pages. diff --git a/apps/frontend/tsconfig.json b/apps/frontend/tsconfig.json new file mode 100644 index 0000000..f4fda14 --- /dev/null +++ b/apps/frontend/tsconfig.json @@ -0,0 +1,27 @@ +{ + "compilerOptions": { + "target": "ES2017", + "lib": ["dom", "dom.iterable", "esnext"], + "allowJs": false, + "skipLibCheck": true, + "strict": true, + "noEmit": true, + "esModuleInterop": true, + "module": "esnext", + "moduleResolution": "bundler", + "resolveJsonModule": true, + "isolatedModules": true, + "jsx": "preserve", + "incremental": true, + "plugins": [ + { + "name": "next" + } + ], + "paths": { + "@/*": ["./src/*"] + } + }, + "include": ["next-env.d.ts", "**/*.ts", "**/*.tsx", ".next/types/**/*.ts"], + "exclude": ["node_modules"] +} diff --git a/apps/runner/Dockerfile b/apps/runner/Dockerfile new file mode 100644 index 0000000..e0665b2 --- /dev/null +++ b/apps/runner/Dockerfile @@ -0,0 +1,12 @@ +FROM python:3.12-slim + +ENV PYTHONDONTWRITEBYTECODE=1 +ENV PYTHONUNBUFFERED=1 + +WORKDIR /app + +COPY apps/runner/src ./src + +EXPOSE 8010 + +CMD ["python", "-m", "src.presentation.main"] diff --git a/apps/runner/requirements.txt b/apps/runner/requirements.txt new file mode 100644 index 0000000..e69de29 diff --git a/apps/runner/src/__init__.py b/apps/runner/src/__init__.py new file mode 100644 index 0000000..d6a16e2 --- /dev/null +++ b/apps/runner/src/__init__.py @@ -0,0 +1 @@ +"""Runner service package.""" diff --git a/apps/runner/src/application/README.md b/apps/runner/src/application/README.md new file mode 100644 index 0000000..8164db1 --- /dev/null +++ b/apps/runner/src/application/README.md @@ -0,0 +1,3 @@ +# Application + +Runner use cases, job orchestration, and command policies belong in this layer. diff --git a/apps/runner/src/application/__init__.py b/apps/runner/src/application/__init__.py new file mode 100644 index 0000000..62502e0 --- /dev/null +++ b/apps/runner/src/application/__init__.py @@ -0,0 +1 @@ +"""Application layer for runner use cases.""" diff --git a/apps/runner/src/domain/README.md b/apps/runner/src/domain/README.md new file mode 100644 index 0000000..58e5923 --- /dev/null +++ b/apps/runner/src/domain/README.md @@ -0,0 +1,3 @@ +# Domain + +Runner job entities and invariants belong in this layer. diff --git a/apps/runner/src/domain/__init__.py b/apps/runner/src/domain/__init__.py new file mode 100644 index 0000000..090ff89 --- /dev/null +++ b/apps/runner/src/domain/__init__.py @@ -0,0 +1 @@ +"""Domain layer for runner job concepts.""" diff --git a/apps/runner/src/infrastructure/README.md b/apps/runner/src/infrastructure/README.md new file mode 100644 index 0000000..3333cc1 --- /dev/null +++ b/apps/runner/src/infrastructure/README.md @@ -0,0 +1,4 @@ +# Infrastructure + +CLI execution, workspace, filesystem, and artifact storage adapters belong in +this layer. diff --git a/apps/runner/src/infrastructure/__init__.py b/apps/runner/src/infrastructure/__init__.py new file mode 100644 index 0000000..329da17 --- /dev/null +++ b/apps/runner/src/infrastructure/__init__.py @@ -0,0 +1 @@ +"""Infrastructure adapters for runner execution.""" diff --git a/apps/runner/src/presentation/README.md b/apps/runner/src/presentation/README.md new file mode 100644 index 0000000..fec74df --- /dev/null +++ b/apps/runner/src/presentation/README.md @@ -0,0 +1,3 @@ +# Presentation + +Runner HTTP routes and API wiring belong in this layer. diff --git a/apps/runner/src/presentation/__init__.py b/apps/runner/src/presentation/__init__.py new file mode 100644 index 0000000..bce5fe9 --- /dev/null +++ b/apps/runner/src/presentation/__init__.py @@ -0,0 +1 @@ +"""Presentation layer for runner HTTP APIs.""" diff --git a/apps/runner/src/presentation/main.py b/apps/runner/src/presentation/main.py new file mode 100644 index 0000000..8b37e1d --- /dev/null +++ b/apps/runner/src/presentation/main.py @@ -0,0 +1,33 @@ +from __future__ import annotations + +import json +import os +from http import HTTPStatus +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + + +class HealthHandler(BaseHTTPRequestHandler): + def do_GET(self) -> None: + if self.path != "/health": + self.send_error(HTTPStatus.NOT_FOUND) + return + + body = json.dumps({"service": "runner", "status": "ok"}).encode("utf-8") + self.send_response(HTTPStatus.OK) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + + def log_message(self, format: str, *args: object) -> None: + return + + +def run() -> None: + port = int(os.environ.get("PORT", "8010")) + server = ThreadingHTTPServer(("0.0.0.0", port), HealthHandler) + server.serve_forever() + + +if __name__ == "__main__": + run() diff --git a/apps/runner/src/shared/README.md b/apps/runner/src/shared/README.md new file mode 100644 index 0000000..cb615e7 --- /dev/null +++ b/apps/runner/src/shared/README.md @@ -0,0 +1,3 @@ +# Shared + +Cross-layer runner primitives belong here. Keep this layer small. diff --git a/apps/runner/src/shared/__init__.py b/apps/runner/src/shared/__init__.py new file mode 100644 index 0000000..a2ca85e --- /dev/null +++ b/apps/runner/src/shared/__init__.py @@ -0,0 +1 @@ +"""Shared runner primitives.""" diff --git a/docker-compose.yml b/docker-compose.yml new file mode 100644 index 0000000..a8d871e --- /dev/null +++ b/docker-compose.yml @@ -0,0 +1,106 @@ +services: + frontend: + build: + context: . + dockerfile: apps/frontend/Dockerfile + environment: + NEXT_TELEMETRY_DISABLED: "1" + BACKEND_URL: http://backend:8000 + ports: + - "${FRONTEND_PORT:-3000}:3000" + depends_on: + backend: + condition: service_started + + backend: + build: + context: . + dockerfile: apps/backend/Dockerfile + environment: + POSTGRES_DSN: postgresql://${POSTGRES_USER:-pipeline}:${POSTGRES_PASSWORD:-pipeline_local}@postgres:5432/${POSTGRES_DB:-pipeline} + REDIS_URL: redis://redis:6379/0 + OBJECT_STORAGE_ENDPOINT: http://minio:9000 + OBJECT_STORAGE_ACCESS_KEY_ID: ${MINIO_ROOT_USER:-minio_local} + OBJECT_STORAGE_SECRET_ACCESS_KEY: ${MINIO_ROOT_PASSWORD:-minio_local_password} + OBJECT_STORAGE_BUCKET: ${OBJECT_STORAGE_BUCKET:-pipeline-local} + OBJECT_STORAGE_REGION: ${OBJECT_STORAGE_REGION:-us-east-1} + ports: + - "${BACKEND_PORT:-8000}:8000" + depends_on: + postgres: + condition: service_healthy + redis: + condition: service_healthy + minio-init: + condition: service_completed_successfully + + runner: + build: + context: . + dockerfile: apps/runner/Dockerfile + environment: + RUNNER_MODE: fake + ports: + - "${RUNNER_PORT:-8010}:8010" + + postgres: + image: postgres:16-alpine + environment: + POSTGRES_USER: ${POSTGRES_USER:-pipeline} + POSTGRES_PASSWORD: ${POSTGRES_PASSWORD:-pipeline_local} + POSTGRES_DB: ${POSTGRES_DB:-pipeline} + ports: + - "${POSTGRES_PORT:-5432}:5432" + volumes: + - postgres-data:/var/lib/postgresql/data + healthcheck: + test: ["CMD-SHELL", "pg_isready -U $${POSTGRES_USER} -d $${POSTGRES_DB}"] + interval: 5s + timeout: 3s + retries: 20 + + redis: + image: redis:7-alpine + ports: + - "${REDIS_PORT:-6379}:6379" + volumes: + - redis-data:/data + healthcheck: + test: ["CMD", "redis-cli", "ping"] + interval: 5s + timeout: 3s + retries: 20 + + minio: + image: minio/minio:latest + command: server /data --console-address ":9001" + environment: + MINIO_ROOT_USER: ${MINIO_ROOT_USER:-minio_local} + MINIO_ROOT_PASSWORD: ${MINIO_ROOT_PASSWORD:-minio_local_password} + ports: + - "${MINIO_API_PORT:-9000}:9000" + - "${MINIO_CONSOLE_PORT:-9001}:9001" + volumes: + - minio-data:/data + + minio-init: + image: minio/mc:latest + depends_on: + minio: + condition: service_started + environment: + MINIO_ROOT_USER: ${MINIO_ROOT_USER:-minio_local} + MINIO_ROOT_PASSWORD: ${MINIO_ROOT_PASSWORD:-minio_local_password} + OBJECT_STORAGE_BUCKET: ${OBJECT_STORAGE_BUCKET:-pipeline-local} + entrypoint: ["/bin/sh", "-c"] + command: + - > + until mc alias set local http://minio:9000 "$$MINIO_ROOT_USER" "$$MINIO_ROOT_PASSWORD"; do + sleep 1; + done; + mc mb --ignore-existing "local/$$OBJECT_STORAGE_BUCKET"; + +volumes: + postgres-data: + redis-data: + minio-data: diff --git a/docs/public-health-interfaces.md b/docs/public-health-interfaces.md new file mode 100644 index 0000000..98ca9ee --- /dev/null +++ b/docs/public-health-interfaces.md @@ -0,0 +1,120 @@ +# Public Health Interfaces + +Task 001 pre-requirement contract. These interfaces are the public test surface +for the dockerized foundation implementation. Tests must exercise only the +Docker Compose stack and HTTP endpoints, not service internals. + +## Local Stack + +The local stack starts from the repository root: + +```sh +docker compose up --build +``` + +The implementation must expose these local ports: + +| Service | URL | +| --- | --- | +| Frontend | `http://localhost:3000` | +| Backend | `http://localhost:8000` | +| Runner | `http://localhost:8010` | + +## Health Endpoints + +### Backend Process Health + +```http +GET http://localhost:8000/health +``` + +Expected response: + +```json +{ + "service": "backend", + "status": "ok" +} +``` + +This endpoint proves that the FastAPI process is serving HTTP. + +### Backend Connectivity Smoke + +```http +GET http://localhost:8000/health/dependencies +``` + +Expected response: + +```json +{ + "service": "backend", + "status": "ok", + "dependencies": { + "postgres": "ok", + "redis": "ok", + "object_storage": "ok" + } +} +``` + +This endpoint is the public smoke surface for backend connectivity to Postgres, +Redis, and S3-compatible object storage. Implementations may add fields, but +must preserve these keys and `ok` statuses when dependencies are reachable. + +### Runner Health + +```http +GET http://localhost:8010/health +``` + +Expected response: + +```json +{ + "service": "runner", + "status": "ok" +} +``` + +### Frontend Health + +```http +GET http://localhost:3000/health +``` + +Expected response: + +```json +{ + "service": "frontend", + "status": "ok" +} +``` + +## Architecture Boundaries + +The frontend application must follow Feature-Sliced Design with these top-level +layers: + +```text +app/ +processes/ +pages/ +widgets/ +features/ +entities/ +shared/ +``` + +The backend and runner services must use these top-level architecture layers: + +```text +domain/ +application/ +infrastructure/ +presentation/ +shared/ +``` + diff --git a/infra/docker/README.md b/infra/docker/README.md new file mode 100644 index 0000000..a73c124 --- /dev/null +++ b/infra/docker/README.md @@ -0,0 +1,5 @@ +# Docker Infrastructure + +Local Docker Compose infrastructure for Task 001 lives at the repository root in +`docker-compose.yml`. This directory is reserved for future service-specific +Docker configuration, seed scripts, and local infra overrides. diff --git a/infra/docker/minio/README.md b/infra/docker/minio/README.md new file mode 100644 index 0000000..38df4b3 --- /dev/null +++ b/infra/docker/minio/README.md @@ -0,0 +1,4 @@ +# MinIO + +The root Docker Compose file starts MinIO and uses the `minio-init` service to +create the local object storage bucket declared by `OBJECT_STORAGE_BUCKET`. diff --git a/infra/docker/postgres/README.md b/infra/docker/postgres/README.md new file mode 100644 index 0000000..8ec8b1f --- /dev/null +++ b/infra/docker/postgres/README.md @@ -0,0 +1,3 @@ +# Postgres + +Reserved for local Postgres initialization files and future migrations wiring. diff --git a/package.json b/package.json new file mode 100644 index 0000000..a046e2b --- /dev/null +++ b/package.json @@ -0,0 +1,12 @@ +{ + "name": "ai-content-pipeline", + "private": true, + "scripts": { + "dev:stack": "docker compose up --build", + "smoke": "bash tests/smoke/public-health.sh" + }, + "workspaces": [ + "apps/frontend", + "packages/shared" + ] +} diff --git a/packages/shared/package.json b/packages/shared/package.json new file mode 100644 index 0000000..0d7e5d1 --- /dev/null +++ b/packages/shared/package.json @@ -0,0 +1,14 @@ +{ + "name": "@pipeline/shared", + "private": true, + "type": "module", + "exports": { + ".": "./src/index.ts" + }, + "scripts": { + "typecheck": "tsc --noEmit" + }, + "devDependencies": { + "typescript": "5.7.2" + } +} diff --git a/packages/shared/src/health.ts b/packages/shared/src/health.ts new file mode 100644 index 0000000..92f2f7f --- /dev/null +++ b/packages/shared/src/health.ts @@ -0,0 +1,13 @@ +export type ServiceName = "backend" | "frontend" | "runner"; + +export type HealthResponse = { + service: ServiceName; + status: "ok"; +}; + +export type BackendDependencyName = "postgres" | "redis" | "object_storage"; + +export type BackendDependenciesHealthResponse = HealthResponse & { + service: "backend"; + dependencies: Record; +}; diff --git a/packages/shared/src/index.ts b/packages/shared/src/index.ts new file mode 100644 index 0000000..ce7d76e --- /dev/null +++ b/packages/shared/src/index.ts @@ -0,0 +1,6 @@ +export type { + BackendDependenciesHealthResponse, + BackendDependencyName, + HealthResponse, + ServiceName, +} from "./health"; diff --git a/packages/shared/tsconfig.json b/packages/shared/tsconfig.json new file mode 100644 index 0000000..1bbd885 --- /dev/null +++ b/packages/shared/tsconfig.json @@ -0,0 +1,12 @@ +{ + "compilerOptions": { + "target": "ES2022", + "module": "ESNext", + "moduleResolution": "Bundler", + "strict": true, + "declaration": true, + "noEmit": true, + "skipLibCheck": true + }, + "include": ["src/**/*.ts"] +} diff --git a/pnpm-workspace.yaml b/pnpm-workspace.yaml new file mode 100644 index 0000000..0ae8db2 --- /dev/null +++ b/pnpm-workspace.yaml @@ -0,0 +1,3 @@ +packages: + - "apps/frontend" + - "packages/shared" diff --git a/tasks/001-dockerized-monorepo-foundation.md b/tasks/001-dockerized-monorepo-foundation.md new file mode 100644 index 0000000..76c2d11 --- /dev/null +++ b/tasks/001-dockerized-monorepo-foundation.md @@ -0,0 +1,102 @@ +# Task 001: Dockerized Monorepo Foundation + +Development description: Create the deployable project foundation for the AI Content Pipeline with backend, frontend, runner, shared contracts, and local infrastructure booting through one Docker Compose command. + +## Implementation Details + +- Create a monorepo layout: + - `apps/backend` for FastAPI. + - `apps/frontend` for Next.js. + - `apps/runner` for the agent runner service. + - `packages/shared` for shared schemas and generated clients. + - `infra/docker` for local infra configuration. +- Add `docker-compose.yml` with services for frontend, backend, runner, Postgres, Redis, and S3-compatible object storage such as MinIO. +- Add health endpoints: + - Backend: `GET /health`. + - Runner: `GET /health`. + - Frontend route: `/health`. +- Add a root developer command such as `make dev` or `pnpm dev:stack` that starts the full stack. +- Add `.env.example` with local-only defaults and no secrets. + +## Public Interface + +- `docker compose up --build` starts the stack. +- `GET http://localhost:/health` returns backend health. +- `GET http://localhost:/health` returns runner health. +- `GET http://localhost:/health` returns frontend health. + +## Acceptance Criteria + +- [x] TDD pre-requirement: before implementation, define the public health interfaces and write one failing stack/health check test first; proceed one red-green-refactor cycle at a time and record evidence in `Result`. +- [x] A clean checkout can build and start the stack with one documented command. +- [x] Backend, frontend, runner, Postgres, Redis, and object storage containers start without manual setup. +- [x] Health checks pass from outside the containers. +- [x] Backend can connect to Postgres, Redis, and object storage using environment variables. +- [x] No real credentials are committed. + +## Verification + +- Run `docker compose up --build`. +- Run health checks against backend, frontend, and runner. +- Run a smoke test that verifies backend connectivity to Postgres, Redis, and object storage. + +## Result + +- Status: Completed; public health smoke is green. +- TDD plan: + - Public contract documented in `docs/public-health-interfaces.md`. + - First vertical Red test added at `tests/smoke/public-health.sh`. + - The test starts the stack with `docker compose -f docker-compose.yml up --build -d`. + - The test verifies only public HTTP surfaces: + - Backend process health: `GET http://localhost:8000/health`. + - Runner health: `GET http://localhost:8010/health`. + - Frontend health: `GET http://localhost:3000/health`. + - Backend dependency smoke: `GET http://localhost:8000/health/dependencies` for Postgres, Redis, and object storage connectivity. + - Implementation agent should make this single test green before adding the next public behavior test. +- Red evidence: + - Command: `bash tests/smoke/public-health.sh`. + - Actual output: + ```text + Starting stack with docker compose... + open /Users/gavrilovdev/tmp/pupline/docker-compose.yml: no such file or directory + ``` + - Result: failed as expected because the Docker Compose stack and service health implementations do not exist yet. +- Green evidence: + - Command: `bash tests/smoke/public-health.sh`. + - Final observed output excerpt: + ```text + Starting stack with docker compose... + ... + #22 106.2 found 0 vulnerabilities + ... + pupline-frontend Built + pupline-runner Built + pupline-backend Built + ... + Container pupline-backend-1 Started + Container pupline-frontend-1 Started + backend health ok + runner health ok + frontend health ok + backend dependencies ok + public health smoke ok + ``` +- Refactor notes: + - Backend implemented as FastAPI with `/health` and `/health/dependencies`. + - Runner health uses a minimal stdlib HTTP service to avoid unnecessary image dependencies for Task 001. + - Frontend uses Next.js App Router under `apps/frontend/src/app`; FSD placeholder layers live under `apps/frontend/src/processes`, `src/pages`, `src/widgets`, `src/features`, `src/entities`, and `src/shared`. + - Frontend pins Next.js `16.2.6` and overrides PostCSS to `8.5.14`; Docker build `npm install` reported `found 0 vulnerabilities`. +- Verification output: + - Command: `python3 -m py_compile apps/backend/src/presentation/main.py apps/backend/src/infrastructure/dependencies.py apps/runner/src/presentation/main.py`. + Output: no output, exit code 0. + - Command: `docker compose config --quiet`. + Output: no output, exit code 0. + - Command: `bash tests/smoke/public-health.sh`. + Output: + ```text + backend health ok + runner health ok + frontend health ok + backend dependencies ok + public health smoke ok + ``` diff --git a/tasks/002-shared-domain-contracts.md b/tasks/002-shared-domain-contracts.md new file mode 100644 index 0000000..a98b373 --- /dev/null +++ b/tasks/002-shared-domain-contracts.md @@ -0,0 +1,53 @@ +# Task 002: Shared Domain Contracts + +Development description: Implement the shared domain schema package that defines workflow statuses, role names, API DTOs, and validation contracts used consistently by backend, frontend, runner, and tests. + +## Implementation Details + +- Define canonical enums for: + - Roles: `ADMIN`, `EDITOR`. + - Article workflow statuses from `ARTICLE_BRIEF_CREATED` through `PUBLISH_COMMIT_CREATED`. + - Agent job statuses and error categories. + - Claim support statuses and risk levels. + - Publishing statuses. +- Define shared request/response schemas for: + - Article create/list/detail. + - Target site config. + - Script/config version. + - Agent job. + - Research artifact manifest. + - Plan, draft, evidence, asset, review, and publish commit summaries. +- Choose one source of truth for validation: + - Recommended implementation: Pydantic models in backend plus generated OpenAPI client/types for frontend. + - Shared package may hold OpenAPI schema snapshots and TypeScript types generated from backend. +- Add schema validation tests for representative valid and invalid payloads. + +## Public Interface + +- Backend exposes OpenAPI JSON with the canonical contracts. +- Frontend imports generated API types instead of duplicating DTO shapes. +- Runner consumes backend contracts for job input/output validation. + +## Acceptance Criteria + +- [ ] TDD pre-requirement: before implementation, write a failing contract test for one externally visible DTO and one invalid payload; proceed one schema at a time and record red-green evidence in `Result`. +- [ ] Workflow status transitions use shared constants rather than string literals spread across apps. +- [ ] Role names are exactly `ADMIN` and `EDITOR`. +- [ ] Generated frontend types match backend OpenAPI. +- [ ] Contract tests fail on missing required fields and invalid enum values. +- [ ] Runner job output schemas can be validated without importing frontend code. + +## Verification + +- Run backend schema tests. +- Generate frontend API types. +- Run a typecheck in frontend against generated types. + +## Result + +- Status: Pending execution. +- TDD plan: To be filled during execution. +- Red evidence: To be filled during execution. +- Green evidence: To be filled during execution. +- Refactor notes: To be filled during execution. +- Verification output: To be filled during execution. diff --git a/tasks/003-postgres-schema-and-seed-data.md b/tasks/003-postgres-schema-and-seed-data.md new file mode 100644 index 0000000..0036c75 --- /dev/null +++ b/tasks/003-postgres-schema-and-seed-data.md @@ -0,0 +1,62 @@ +# Task 003: Postgres Schema And Seed Data + +Development description: Create the initial relational schema, migrations, indexes, and seed data for articles, sites, workflow state, jobs, research manifests, publishing versions, and audit events. + +## Implementation Details + +- Add migration tooling for the FastAPI backend, such as Alembic. +- Create tables: + - `users` + - `target_sites` + - `script_config_versions` + - `articles` + - `boundary_questions` + - `article_plans` + - `plan_sections` + - `evidence_items` + - `claims` + - `article_drafts` + - `assets` + - `workflow_events` + - `agent_jobs` + - `research_run_manifests` + - `publish_commits` + - `prompt_versions` +- Store article/domain status in domain tables as authoritative state. +- Store LangGraph checkpoints separately from domain tables if checkpointing is implemented in this task. +- Seed: + - One Admin user. + - One Editor user. + - One Git-backed Next target site. + - One active publishing YAML/script config version. +- Add indexes for article dashboard queries, job queues, workflow timeline, and site lookup by slug. + +## Public Interface + +- `alembic upgrade head` creates the schema. +- Backend repository methods can create/read seeded target sites and users. + +## Acceptance Criteria + +- [ ] TDD pre-requirement: before implementing migrations, write one failing integration test through repository/public backend interfaces for seeded site lookup; add further tests one behavior at a time and record red-green evidence in `Result`. +- [ ] Migrations run cleanly on an empty Postgres database. +- [ ] Seed data is idempotent. +- [ ] Article/domain tables hold authoritative article statuses. +- [ ] Script/config versions store author, timestamp, diff, rollback target, activation timestamp, YAML, and transform script. +- [ ] Research manifests store S3 prefixes, source URLs, object keys, content hashes, artifact types, and metadata references. +- [ ] Publish commits store repository URL, branch, commit SHA, content bundle manifest, and status. + +## Verification + +- Run migration tests against a disposable Postgres database. +- Run seed command twice and verify no duplicates. +- Run repository integration tests. + +## Result + +- Status: Pending execution. +- TDD plan: To be filled during execution. +- Red evidence: To be filled during execution. +- Green evidence: To be filled during execution. +- Refactor notes: To be filled during execution. +- Verification output: To be filled during execution. diff --git a/tasks/004-auth-and-admin-editor-roles.md b/tasks/004-auth-and-admin-editor-roles.md new file mode 100644 index 0000000..6c2f913 --- /dev/null +++ b/tasks/004-auth-and-admin-editor-roles.md @@ -0,0 +1,48 @@ +# Task 004: Auth And Admin/Editor Roles + +Development description: Implement simple internal authentication and role-based authorization for the two v1 roles: Admin and Editor. + +## Implementation Details + +- Add a local-demo authentication mode suitable for Docker Compose: + - Header-based user selection for local demo, or + - Session login with seeded users. +- Enforce role checks at backend endpoint boundaries. +- Role capabilities: + - Admin can edit target sites, publishing YAML/scripts, prompt versions, and runner profiles. + - Editor can create and run article pipelines, edit intermediate outputs, approve content, and create publish commits. +- Add frontend role-aware navigation: + - Admin sees site configuration and script versioning screens. + - Editor sees pipeline execution and review screens. +- Ensure authorization failures return stable `403` responses. + +## Public Interface + +- Backend identifies current user and role for each request. +- Frontend can fetch `GET /api/me`. +- Protected endpoints consistently allow or deny Admin/Editor actions. + +## Acceptance Criteria + +- [ ] TDD pre-requirement: before implementation, write one failing public API authorization test for an Editor attempting an Admin-only action; proceed one permission behavior at a time and record red-green evidence in `Result`. +- [ ] `GET /api/me` returns the active user and role. +- [ ] Admin can create/update site config and script versions. +- [ ] Editor cannot create/update site config or script versions. +- [ ] Editor can create articles and perform review actions. +- [ ] Unauthorized requests are rejected consistently. +- [ ] Frontend hides Admin-only navigation for Editors. + +## Verification + +- Run backend authorization tests. +- Run frontend role rendering tests. +- Manually verify Admin and Editor demo sessions in Docker Compose. + +## Result + +- Status: Pending execution. +- TDD plan: To be filled during execution. +- Red evidence: To be filled during execution. +- Green evidence: To be filled during execution. +- Refactor notes: To be filled during execution. +- Verification output: To be filled during execution. diff --git a/tasks/005-article-intake-dashboard-detail.md b/tasks/005-article-intake-dashboard-detail.md new file mode 100644 index 0000000..05aa252 --- /dev/null +++ b/tasks/005-article-intake-dashboard-detail.md @@ -0,0 +1,54 @@ +# Task 005: Article Intake, Dashboard, And Detail Shell + +Development description: Build the first user-visible workflow slice where an Editor creates an article brief, sees it on the dashboard, and opens an article detail page with workflow history. + +## Implementation Details + +- Backend endpoints: + - `POST /api/articles` + - `GET /api/articles` + - `GET /api/articles/{article_id}` +- Article creation requires: + - `brief_description` + - `target_site_id` + - optional `content_type` + - optional `primary_keyword` +- Creation behavior: + - Creates article with `ARTICLE_BRIEF_CREATED`. + - Attaches target site. + - Writes `ARTICLE_CREATED` workflow event. +- Frontend screens: + - Dashboard with article title/working title, target website, status, assigned editor, last updated, next required action, and publishing status. + - New article form. + - Article detail shell with status summary and timeline. + +## Public Interface + +- Editor submits a brief in the UI. +- Dashboard updates with the created article. +- Article detail page shows the workflow event timeline. + +## Acceptance Criteria + +- [ ] TDD pre-requirement: before implementation, write one failing API behavior test for creating an article and seeing `ARTICLE_BRIEF_CREATED`; add UI tests after API green and record evidence in `Result`. +- [ ] Editor can create an article brief for a configured target site. +- [ ] Created article appears in `GET /api/articles`. +- [ ] Article detail includes target site summary and workflow history. +- [ ] Article creation writes a workflow event with actor and timestamp. +- [ ] Invalid target site returns a validation error. +- [ ] Frontend form shows validation errors without losing typed brief text. + +## Verification + +- Run backend article API tests. +- Run frontend component/page tests for dashboard and form. +- Run a Docker Compose smoke flow from UI brief creation to article detail. + +## Result + +- Status: Pending execution. +- TDD plan: To be filled during execution. +- Red evidence: To be filled during execution. +- Green evidence: To be filled during execution. +- Refactor notes: To be filled during execution. +- Verification output: To be filled during execution. diff --git a/tasks/006-site-config-and-script-versioning.md b/tasks/006-site-config-and-script-versioning.md new file mode 100644 index 0000000..7ee968e --- /dev/null +++ b/tasks/006-site-config-and-script-versioning.md @@ -0,0 +1,62 @@ +# Task 006: Site Config And Publishing Script Versioning + +Development description: Build Admin-managed target site configuration with versioned publishing YAML and transformation scripts that can be activated, audited, and rolled back. + +## Implementation Details + +- Backend endpoints: + - `POST /api/sites` + - `GET /api/sites` + - `GET /api/sites/{site_id}` + - `PATCH /api/sites/{site_id}` + - `GET /api/sites/{site_id}/publishing-config/versions` + - `POST /api/sites/{site_id}/publishing-config/versions` + - `POST /api/sites/{site_id}/publishing-config/versions/{version_id}/activate` + - `POST /api/sites/{site_id}/publishing-config/versions/{version_id}/rollback` +- Site config fields: + - Repository URL. + - Production branch. + - Content format and path templates. + - Asset path template. + - Frontmatter mapping. + - Generic preview renderer selection. + - Editorial, SEO, source, visual, and publishing rules. +- Script/config versions store: + - YAML config. + - Transform script text. + - Author. + - Timestamp. + - Diff from previous version. + - Rollback target. + - Activation timestamp. +- No second Admin approval is required in v1. + +## Public Interface + +- Admin can create and activate a site publishing config version. +- Editor can read active site config but cannot mutate it. + +## Acceptance Criteria + +- [ ] TDD pre-requirement: before implementation, write one failing Admin API test for creating a new publishing config version and one failing Editor denial test; proceed one behavior at a time and record evidence in `Result`. +- [ ] Admin can create a target site with Git-backed publishing fields. +- [ ] Admin can create a new YAML/script version and activate it. +- [ ] Version creation records author, timestamp, diff, and rollback target. +- [ ] Rollback activates a previous version and writes an audit event. +- [ ] Editor can view active site config but cannot create, activate, or roll back versions. +- [ ] Frontend Admin UI displays version history and active version. + +## Verification + +- Run backend site config API tests. +- Run frontend Admin site config tests. +- Manually create, activate, and roll back a site config in Docker Compose. + +## Result + +- Status: Pending execution. +- TDD plan: To be filled during execution. +- Red evidence: To be filled during execution. +- Green evidence: To be filled during execution. +- Refactor notes: To be filled during execution. +- Verification output: To be filled during execution. diff --git a/tasks/007-agent-job-queue-and-runner.md b/tasks/007-agent-job-queue-and-runner.md new file mode 100644 index 0000000..433ed61 --- /dev/null +++ b/tasks/007-agent-job-queue-and-runner.md @@ -0,0 +1,57 @@ +# Task 007: Agent Job Queue And Runner MVP + +Development description: Implement the durable job path from backend to runner, including queued jobs, isolated workspaces, CLI command execution, log capture, output validation, retry, and cancellation basics. + +## Implementation Details + +- Backend: + - `POST /api/agent-jobs/test-codex` + - `GET /api/agent-jobs` + - `GET /api/agent-jobs/{job_id}` + - `POST /api/agent-jobs/{job_id}/retry` + - `POST /api/agent-jobs/{job_id}/cancel` +- Queue: + - Use Redis-backed Celery, Dramatiq, or RQ. + - Persist `QUEUED`, `RUNNING`, `SUCCEEDED`, `FAILED`, `CANCELLED`. +- Runner: + - Creates one workspace per job. + - Writes structured input files. + - Executes allowed command. + - Captures stdout, stderr, exit code, duration. + - Validates output JSON against the expected schema. + - Uploads job artifacts to object storage where appropriate. +- Test Codex job: + - Must support a demo-safe fake runner mode for CI/local tests. + - Real Codex CLI execution is configurable for environments with runner auth. + +## Public Interface + +- Admin can enqueue a test runner job. +- UI can show job status and logs. +- Failed jobs can be retried where allowed. + +## Acceptance Criteria + +- [ ] TDD pre-requirement: before implementation, write one failing backend-to-fake-runner integration test through public job APIs; implement minimal queue/runner behavior to make it green and record evidence in `Result`. +- [ ] Backend can create and persist an agent job. +- [ ] Runner claims a queued job and marks it running. +- [ ] Runner stores stdout, stderr, exit code, duration, and final status. +- [ ] Runner validates output JSON and marks schema failures as `FAILED_SCHEMA_VALIDATION`. +- [ ] Cancelled jobs do not update article domain state. +- [ ] Retry creates a new attempt or new job with traceable parent metadata. +- [ ] Demo stack works without real Codex credentials by using fake runner mode. + +## Verification + +- Run backend/runner integration tests with fake runner. +- Run a Docker Compose test job from Admin UI or API. +- Verify logs and status in UI. + +## Result + +- Status: Pending execution. +- TDD plan: To be filled during execution. +- Red evidence: To be filled during execution. +- Green evidence: To be filled during execution. +- Refactor notes: To be filled during execution. +- Verification output: To be filled during execution. diff --git a/tasks/008-boundary-questions-loop.md b/tasks/008-boundary-questions-loop.md new file mode 100644 index 0000000..252958e --- /dev/null +++ b/tasks/008-boundary-questions-loop.md @@ -0,0 +1,61 @@ +# Task 008: Boundary Questions Loop + +Development description: Implement generation, editing, partial saving, and submission of boundary questions before plan generation. + +## Implementation Details + +- Backend endpoints: + - `POST /api/articles/{article_id}/boundary-questions/generate` + - `GET /api/articles/{article_id}/boundary-questions` + - `PATCH /api/articles/{article_id}/boundary-questions/{question_id}` + - `POST /api/articles/{article_id}/boundary-questions/submit` +- Generate 5-10 questions covering: + - Audience. + - Purpose. + - Reader outcome. + - Depth. + - Tone. + - Excluded topics. + - Primary keyword. + - Competitor angle. + - Evidence standard. + - Visual expectations. +- Use agent job path with fake runner fixtures for tests. +- Store editable questions and answers. +- Block plan generation until all required questions are answered and submitted. +- Frontend: + - Boundary questions screen. + - Partial save. + - Required answer validation. + +## Public Interface + +- Editor generates questions for an article. +- Editor edits answers and submits them. +- Article moves to `BOUNDARY_ANSWERS_SUBMITTED`. + +## Acceptance Criteria + +- [ ] TDD pre-requirement: before implementation, write one failing API behavior test for required unanswered questions blocking submission; proceed through generate/save/submit behavior one cycle at a time and record evidence in `Result`. +- [ ] Questions are generated from article brief plus target site config. +- [ ] Required categories are represented. +- [ ] Editor can save partial answers. +- [ ] Required unanswered questions block submission. +- [ ] Submission writes workflow event and updates article status. +- [ ] Plan generation remains blocked until boundary answers are submitted. +- [ ] Frontend displays required/optional state clearly. + +## Verification + +- Run boundary question API tests. +- Run frontend page tests. +- Run a Docker Compose smoke flow from article creation to boundary answer submission. + +## Result + +- Status: Pending execution. +- TDD plan: To be filled during execution. +- Red evidence: To be filled during execution. +- Green evidence: To be filled during execution. +- Refactor notes: To be filled during execution. +- Verification output: To be filled during execution. diff --git a/tasks/009-plan-generation-and-review.md b/tasks/009-plan-generation-and-review.md new file mode 100644 index 0000000..f2c66a1 --- /dev/null +++ b/tasks/009-plan-generation-and-review.md @@ -0,0 +1,64 @@ +# Task 009: Plan Generation And Review Gate + +Development description: Implement article plan generation, immutable plan versioning, direct plan edits, revision requests, and final plan approval before research. + +## Implementation Details + +- Backend endpoints: + - `POST /api/articles/{article_id}/plan/generate` + - `GET /api/articles/{article_id}/plans` + - `GET /api/articles/{article_id}/plans/{plan_id}` + - `PATCH /api/articles/{article_id}/plans/{plan_id}` + - `POST /api/articles/{article_id}/plans/{plan_id}/approve` + - `POST /api/articles/{article_id}/plans/{plan_id}/request-revision` +- Plan generation uses: + - Brief. + - Boundary answers. + - Target site config. + - Output schema. +- Plan must include: + - Title options. + - Recommended title. + - Reader persona. + - Search intent. + - Thesis. + - At least four sections. + - Claims to prove. + - Evidence needs. + - Visual needs. + - SEO notes. + - Risks. +- Edits create new immutable versions rather than mutating approval-relevant content in place. +- Research cannot start until plan is approved. + +## Public Interface + +- Editor generates a plan. +- Editor edits, requests revision, or approves the plan. +- Approved plan becomes the production contract for research. + +## Acceptance Criteria + +- [ ] TDD pre-requirement: before implementation, write one failing public API test showing research cannot start before plan approval; implement generation/review behavior in vertical cycles and record evidence in `Result`. +- [ ] Plan generation creates version 1 and moves article to `PLAN_REVIEW_REQUIRED`. +- [ ] Plan has at least four sections with purpose, key points, and evidence needs. +- [ ] Direct plan edit creates a new immutable version. +- [ ] Revision request writes workflow event and returns workflow to plan generation/revision. +- [ ] Approval writes workflow event with actor, timestamp, and exact plan version. +- [ ] Previous plan versions remain accessible. +- [ ] Frontend supports approve, request revision, direct edit, section notes, source requirements, excluded sources, visual requirements, tone, audience, and SEO keyword changes. + +## Verification + +- Run plan API integration tests with fake runner output. +- Run frontend plan review tests. +- Run a Docker Compose smoke flow through plan approval. + +## Result + +- Status: Pending execution. +- TDD plan: To be filled during execution. +- Red evidence: To be filled during execution. +- Green evidence: To be filled during execution. +- Refactor notes: To be filled during execution. +- Verification output: To be filled during execution. diff --git a/tasks/010-research-fetch-and-s3-manifest.md b/tasks/010-research-fetch-and-s3-manifest.md new file mode 100644 index 0000000..988cb8f --- /dev/null +++ b/tasks/010-research-fetch-and-s3-manifest.md @@ -0,0 +1,61 @@ +# Task 010: Research Fetch And S3 Manifest + +Development description: Implement explicit auditable web research with source retrieval metadata, source-section snapshots, object-storage uploads, and Postgres manifest records. + +## Implementation Details + +- Backend endpoint: + - `POST /api/articles/{article_id}/research/start` + - `GET /api/articles/{article_id}/research` +- Research starts only after approved plan. +- Research uses an explicit search/fetch script, not only free-form agent browsing. +- For discovered sources, store in object storage: + - Relevant content sections. + - Page metadata. + - Search path to the article. + - Agent/source-selection criteria. + - Page `` metadata. + - Server IP address. + - Domain WHOIS owner when available. +- In Postgres store only: + - Research run manifest. + - S3/object keys. + - Source URLs. + - Content hashes. + - Artifact types. + - Summary metadata required for UI and validation. +- Retention is indefinite for v1. +- Provide deterministic fake search/fetch fixtures for tests and demo. + +## Public Interface + +- Editor starts research for an approved article plan. +- UI shows research run status and discovered source summary. +- Backend exposes manifest-linked evidence summaries. + +## Acceptance Criteria + +- [ ] TDD pre-requirement: before implementation, write one failing behavior test that starts research from an approved plan and verifies a manifest with object keys is produced; proceed one research behavior at a time and record evidence in `Result`. +- [ ] Research cannot start before plan approval. +- [ ] Research run creates an agent/job record and article status `RESEARCH_RUNNING`. +- [ ] Source-section artifacts are uploaded to object storage. +- [ ] Manifest records include object keys, hashes, URLs, artifact types, and metadata references. +- [ ] Re-running research creates a new manifest instead of overwriting old artifacts. +- [ ] UI can show source URL, title, domain, type, summary, retrieval timestamp, and artifact link. +- [ ] Demo stack can run with deterministic fake research results. + +## Verification + +- Run research integration tests with fake search/fetch. +- Verify object storage contains source-section artifact files. +- Verify Postgres manifest records point to those files. +- Run Docker Compose research smoke flow. + +## Result + +- Status: Pending execution. +- TDD plan: To be filled during execution. +- Red evidence: To be filled during execution. +- Green evidence: To be filled during execution. +- Refactor notes: To be filled during execution. +- Verification output: To be filled during execution. diff --git a/tasks/011-evidence-matrix-and-claim-gate.md b/tasks/011-evidence-matrix-and-claim-gate.md new file mode 100644 index 0000000..26857b3 --- /dev/null +++ b/tasks/011-evidence-matrix-and-claim-gate.md @@ -0,0 +1,57 @@ +# Task 011: Evidence Matrix And Claim Gate + +Development description: Build claim extraction, claim-to-evidence mapping, unsupported-claim visibility, insufficient-evidence plan revision, and high-risk unsupported-claim blocking. + +## Implementation Details + +- Backend endpoints: + - `GET /api/articles/{article_id}/evidence` + - `PATCH /api/articles/{article_id}/evidence/{evidence_id}` + - Claim review operations under article evidence routes. +- Evidence matrix maps: + - Plan sections. + - Claims. + - Evidence items. + - Source quality score. + - Support status. + - Risk level. +- If research cannot find enough acceptable evidence for the approved plan: + - Move workflow to `PLAN_REVISION_REQUIRED`. + - Do not allow drafting. + - Show missing evidence reasons to Editor. +- High-risk unsupported claims: + - Are visible in review screens. + - Block final approval until resolved. +- Editor can approve/reject evidence and add/remove evidence manually. + +## Public Interface + +- Editor views evidence by section and claim. +- Editor can filter unsupported claims. +- Workflow refuses to draft if evidence is structurally insufficient. + +## Acceptance Criteria + +- [ ] TDD pre-requirement: before implementation, write one failing behavior test showing insufficient evidence returns the article to `PLAN_REVISION_REQUIRED`; proceed one claim/evidence behavior at a time and record evidence in `Result`. +- [ ] Every factual claim is mapped to at least one evidence item or marked unsupported. +- [ ] Unsupported claims are visible by section. +- [ ] High-risk unsupported claims block final approval. +- [ ] Insufficient acceptable evidence blocks draft production and triggers plan revision. +- [ ] Editor can approve, reject, add, and remove evidence. +- [ ] Evidence UI supports section and unsupported-claim filters. +- [ ] Evidence item retains URL, title, source type, summary, quality score, retrieval timestamp, and manifest reference. + +## Verification + +- Run evidence matrix API tests. +- Run frontend evidence UI tests. +- Run a smoke flow with both sufficient and insufficient fake research fixtures. + +## Result + +- Status: Pending execution. +- TDD plan: To be filled during execution. +- Red evidence: To be filled during execution. +- Green evidence: To be filled during execution. +- Refactor notes: To be filled during execution. +- Verification output: To be filled during execution. diff --git a/tasks/012-parallel-production-jobs.md b/tasks/012-parallel-production-jobs.md new file mode 100644 index 0000000..6601c28 --- /dev/null +++ b/tasks/012-parallel-production-jobs.md @@ -0,0 +1,57 @@ +# Task 012: Parallel Production Jobs + +Development description: Implement parallel section scaffolding, FAQ, SEO brief, tone guide, table specs, diagram specs, and hero prompt jobs after the evidence matrix is ready. + +## Implementation Details + +- Backend orchestration creates independent jobs for: + - One section scaffold per approved plan section. + - Hero image prompt. + - Diagram specifications. + - Table specifications. + - FAQ block. + - SEO metadata. + - Internal link suggestions. + - Tone guide. +- Each section scaffold output includes: + - `section_id` + - `heading` + - `draft_markdown` + - `used_evidence_ids` + - `unsupported_claims` + - `suggested_visuals` +- Failed section jobs can be retried independently. +- No scaffold may silently introduce unsupported factual claims. +- Use fake runner fixtures for deterministic tests and demo. + +## Public Interface + +- Editor starts production after evidence matrix is ready. +- UI shows parallel job status per production artifact. +- Editor can retry failed section jobs individually. + +## Acceptance Criteria + +- [ ] TDD pre-requirement: before implementation, write one failing behavior test that starts production and creates one job per plan section; proceed one artifact behavior at a time and record evidence in `Result`. +- [ ] Production cannot start before evidence matrix is ready. +- [ ] One section scaffold job is created per approved plan section. +- [ ] Jobs run independently and can fail independently. +- [ ] Retrying one failed section does not rerun successful sections. +- [ ] Each scaffold lists used evidence IDs. +- [ ] Unsupported claims introduced during scaffolding are captured and shown. +- [ ] UI displays per-artifact job state. + +## Verification + +- Run orchestration tests with fake runner. +- Run retry behavior tests. +- Run Docker Compose smoke flow from evidence-ready to production artifacts ready. + +## Result + +- Status: Pending execution. +- TDD plan: To be filled during execution. +- Red evidence: To be filled during execution. +- Green evidence: To be filled during execution. +- Refactor notes: To be filled during execution. +- Verification output: To be filled during execution. diff --git a/tasks/013-draft-assembly-editor-preview.md b/tasks/013-draft-assembly-editor-preview.md new file mode 100644 index 0000000..24a98da --- /dev/null +++ b/tasks/013-draft-assembly-editor-preview.md @@ -0,0 +1,59 @@ +# Task 013: Draft Assembly, Editor, And Generic Preview + +Development description: Assemble section scaffolds into a canonical Markdown/MDX draft, provide versioned draft editing, and render a generic rich preview for editor review. + +## Implementation Details + +- Backend endpoints: + - `POST /api/articles/{article_id}/draft/assemble` + - `GET /api/articles/{article_id}/drafts` + - `GET /api/articles/{article_id}/drafts/{draft_id}` + - `PATCH /api/articles/{article_id}/drafts/{draft_id}` +- Markdown is the canonical editable draft format. +- Draft assembly includes: + - Title. + - Meta title. + - Meta description. + - Body Markdown/MDX. + - FAQ block when applicable. + - Visual placeholders. + - Evidence references. + - Unsupported claim warnings. +- Approval-relevant draft edits create new immutable versions. +- Frontend includes: + - Draft editor. + - Generic Markdown/MDX preview. + - Version selector/history. + - Unsupported-claim warnings near content. +- The preview is for editor UX and content-shape validation, not target-site build/runtime validation. + +## Public Interface + +- Editor assembles a draft after production artifacts are ready. +- Editor edits Markdown and metadata. +- Editor views a generic preview. + +## Acceptance Criteria + +- [ ] TDD pre-requirement: before implementation, write one failing behavior test that assembles a draft from successful section scaffolds through the public API; proceed one draft behavior at a time and record evidence in `Result`. +- [ ] Draft assembly includes all planned sections in order. +- [ ] Draft includes metadata, FAQ, visual placeholders, evidence references, and unsupported claim warnings. +- [ ] Draft version 1 is immutable after approval-relevant edits; edits create a new version. +- [ ] Generic Markdown/MDX preview renders headings, links, tables, images/placeholders, and FAQ. +- [ ] Editor can compare or select draft versions. +- [ ] Draft assembly fails clearly if required section scaffolds are missing. + +## Verification + +- Run draft assembly API tests. +- Run frontend editor/preview tests. +- Run Docker Compose smoke flow from production artifacts to previewed draft. + +## Result + +- Status: Pending execution. +- TDD plan: To be filled during execution. +- Red evidence: To be filled during execution. +- Green evidence: To be filled during execution. +- Refactor notes: To be filled during execution. +- Verification output: To be filled during execution. diff --git a/tasks/014-assets-and-media-library.md b/tasks/014-assets-and-media-library.md new file mode 100644 index 0000000..5083c64 --- /dev/null +++ b/tasks/014-assets-and-media-library.md @@ -0,0 +1,67 @@ +# Task 014: Assets And Media Library + +Development description: Implement asset specifications, generated or uploaded asset records, object-storage file handling, approval/rejection, replacement, and a media library view. + +## Implementation Details + +- Backend endpoints: + - `POST /api/articles/{article_id}/assets/generate-specs` + - `GET /api/articles/{article_id}/assets` + - `PATCH /api/articles/{article_id}/assets/{asset_id}` + - `POST /api/articles/{article_id}/assets/{asset_id}/approve` + - Upload/replacement endpoint for asset files. +- Asset types: + - Hero image. + - Inline diagram. + - Table. + - Flowchart. + - Comparison matrix. + - Architecture diagram. + - Inline image. +- Asset record fields: + - Title. + - Type. + - Prompt or diagram/table code. + - File URL/object key. + - Alt text. + - Caption. + - Status. + - Linked section/article. +- Frontend: + - Asset review page. + - Media library. + - Replacement upload. + - Approval/rejection controls. + +## Public Interface + +- Editor reviews generated asset specs. +- Editor approves, rejects, or replaces assets. +- Approved assets are included in draft/publish bundle. + +## Acceptance Criteria + +- [ ] TDD pre-requirement: before implementation, write one failing public API test for approving an asset and seeing it become available to the article; proceed one asset behavior at a time and record evidence in `Result`. +- [ ] Asset specs are generated and linked to article or section. +- [ ] File uploads store objects in object storage and persist object keys. +- [ ] Asset approval writes workflow event. +- [ ] Rejected assets are excluded from publish bundle. +- [ ] Replacement preserves audit history. +- [ ] Media library lists article assets with status, type, title, preview/file link, alt text, and caption. +- [ ] Draft preview uses approved asset references where available. + +## Verification + +- Run asset API tests. +- Run object-storage upload integration tests. +- Run frontend media library tests. +- Run smoke flow approving and replacing an asset. + +## Result + +- Status: Pending execution. +- TDD plan: To be filled during execution. +- Red evidence: To be filled during execution. +- Green evidence: To be filled during execution. +- Refactor notes: To be filled during execution. +- Verification output: To be filled during execution. diff --git a/tasks/015-seo-and-language-review.md b/tasks/015-seo-and-language-review.md new file mode 100644 index 0000000..6f94e76 --- /dev/null +++ b/tasks/015-seo-and-language-review.md @@ -0,0 +1,72 @@ +# Task 015: SEO And Language Review + +Development description: Implement SEO and linguistic review jobs, issue reports, editor accept/reject/edit flows, and versioned application of accepted suggestions. + +## Implementation Details + +- Backend endpoints: + - `POST /api/articles/{article_id}/seo/review` + - `GET /api/articles/{article_id}/seo/report` + - `POST /api/articles/{article_id}/language/review` + - `GET /api/articles/{article_id}/language/report` + - endpoints to accept/reject/edit individual suggestions. +- SEO checks: + - Title length. + - Meta title length. + - Meta description length. + - H1/H2 structure. + - Keyword placement. + - Internal links. + - External citations. + - Schema type. + - Slug. + - FAQ eligibility. + - Image alt text. + - Readability. + - Duplicate headings. +- Language checks: + - Grammar. + - Spelling. + - Sentence length. + - Passive voice. + - Jargon density. + - Brand tone. + - Forbidden phrases. + - Repetition. + - Weak claims. + - Unsupported certainty. + - CTA consistency. +- Accepted suggestions create a new draft version when they change approval-relevant content. + +## Public Interface + +- Editor runs SEO and language review. +- Editor accepts, rejects, or edits suggestions. +- UI shows issues with severity and locations. + +## Acceptance Criteria + +- [ ] TDD pre-requirement: before implementation, write one failing behavior test that runs an SEO review and returns a visible issue through the public API; proceed one review behavior at a time and record evidence in `Result`. +- [ ] SEO report includes score, issues, suggested fixes, recommended slug/title, and schema JSON. +- [ ] Language report includes severity, location, message, and suggested rewrite. +- [ ] Editor can accept, reject, or edit each suggestion. +- [ ] Accepted content changes create a new immutable draft version. +- [ ] Target-site SEO and tone rules override global defaults. +- [ ] Final review can show unresolved SEO/language issues. +- [ ] Fake review runner produces deterministic demo reports. + +## Verification + +- Run review API tests. +- Run draft-versioning tests for accepted suggestions. +- Run frontend review UI tests. +- Run Docker Compose smoke flow from draft to SEO/language reports. + +## Result + +- Status: Pending execution. +- TDD plan: To be filled during execution. +- Red evidence: To be filled during execution. +- Green evidence: To be filled during execution. +- Refactor notes: To be filled during execution. +- Verification output: To be filled during execution. diff --git a/tasks/016-final-approval-gate.md b/tasks/016-final-approval-gate.md new file mode 100644 index 0000000..34ce8ed --- /dev/null +++ b/tasks/016-final-approval-gate.md @@ -0,0 +1,59 @@ +# Task 016: Final Approval Gate + +Development description: Implement the final review gate that prevents publishing until the editor has approved the exact draft version, assets, evidence state, metadata, and publishing settings. + +## Implementation Details + +- Backend endpoints: + - `POST /api/articles/{article_id}/final-approval` + - `POST /api/articles/{article_id}/final-revision-request` +- Final approval checklist: + - Plan followed. + - Evidence reviewed. + - No high-risk unsupported claims remain. + - SEO metadata approved. + - Images/assets approved. + - Tables and diagrams approved. + - Internal links approved. + - Frontmatter fields selected. + - Content path selected. + - Author selected. + - Publishing mode selected. +- Approval records: + - Actor. + - Timestamp. + - Exact draft version. + - Selected publishing settings. +- Revision request returns workflow to `FINAL_REVISION_REQUIRED`. + +## Public Interface + +- Editor sees final review checklist. +- Editor cannot approve while hard blockers remain. +- Approved article moves to `PUBLISH_DRY_RUN_REQUIRED`. + +## Acceptance Criteria + +- [ ] TDD pre-requirement: before implementation, write one failing behavior test proving high-risk unsupported claims block final approval; proceed one checklist behavior at a time and record evidence in `Result`. +- [ ] Final approval requires an exact draft version. +- [ ] High-risk unsupported claims block approval. +- [ ] Missing required publishing settings block approval. +- [ ] Unapproved required assets block approval. +- [ ] Approval writes a workflow event with actor, timestamp, draft version, and publishing settings. +- [ ] Revision request writes event and moves article to `FINAL_REVISION_REQUIRED`. +- [ ] UI shows blockers and completed checklist items. + +## Verification + +- Run final approval API tests. +- Run frontend final review tests. +- Run smoke flow showing blocked and successful approval paths. + +## Result + +- Status: Pending execution. +- TDD plan: To be filled during execution. +- Red evidence: To be filled during execution. +- Green evidence: To be filled during execution. +- Refactor notes: To be filled during execution. +- Verification output: To be filled during execution. diff --git a/tasks/017-git-publishing-dry-run-and-commit.md b/tasks/017-git-publishing-dry-run-and-commit.md new file mode 100644 index 0000000..c4a67ec --- /dev/null +++ b/tasks/017-git-publishing-dry-run-and-commit.md @@ -0,0 +1,61 @@ +# Task 017: Git Publishing Dry Run And Commit + +Development description: Implement the Git-backed publishing path that builds a content bundle, runs best-effort generic Markdown/MDX dry-run validation, executes Admin-defined transforms on the runner host, and commits directly to the configured production branch. + +## Implementation Details + +- Backend endpoints: + - `POST /api/articles/{article_id}/publishing/dry-run` + - `POST /api/articles/{article_id}/publishing/create-commit` + - `GET /api/articles/{article_id}/publishing/status` + - `GET /api/articles/{article_id}/publishing/commits` +- Content bundle includes: + - Markdown/MDX article file. + - Frontmatter JSON/YAML according to site mapping. + - Referenced approved asset files. + - Bundle manifest. +- Publishing behavior: + - Materialize active versioned YAML/script config into checked-out site repo workspace. + - Run transform script directly on runner host inside site repo workspace. + - Run generic Markdown/MDX dry-run preview validation, not target-site build. + - Commit directly to configured production branch. + - Rely on target repo CI/CD after push. + - Primary done state is `PUBLISH_COMMIT_CREATED`. +- Failure behavior: + - Non-fast-forward push or conflict fails publish step. + - No automatic rebase. + - If delayed deployment verification fails, alert Admin/Editor and leave production commit in place. + +## Public Interface + +- Editor runs publishing dry run after final approval. +- Editor creates publish commit after dry run passes. +- UI shows commit SHA, branch, repository, and opportunistic deployment status. + +## Acceptance Criteria + +- [ ] TDD pre-requirement: before implementation, write one failing integration test against a temporary local Git repository proving a final-approved article can create a commit; add dry-run and failure tests one behavior at a time and record evidence in `Result`. +- [ ] Publishing cannot dry-run before final approval. +- [ ] Dry run fails if generic Markdown/MDX content-shape validation fails. +- [ ] Publish commit cannot run until dry run passes. +- [ ] Content bundle uses site-configured path templates and frontmatter mapping. +- [ ] Active YAML/script config version is included in publish logs/manifest. +- [ ] Publish writes commit to configured production branch in a test repository. +- [ ] Non-fast-forward push or conflict fails without automatic rebase. +- [ ] Publish commit record stores repository URL, branch, commit SHA, bundle manifest, and status. +- [ ] UI clearly labels validation as best-effort content-shape validation only. + +## Verification + +- Run publishing integration tests using a local bare Git repository. +- Run generic preview validation tests. +- Run Docker Compose smoke flow from final approval to publish commit against a demo site repo. + +## Result + +- Status: Pending execution. +- TDD plan: To be filled during execution. +- Red evidence: To be filled during execution. +- Green evidence: To be filled during execution. +- Refactor notes: To be filled during execution. +- Verification output: To be filled during execution. diff --git a/tasks/018-observability-retry-cancel-audit.md b/tasks/018-observability-retry-cancel-audit.md new file mode 100644 index 0000000..0b231f1 --- /dev/null +++ b/tasks/018-observability-retry-cancel-audit.md @@ -0,0 +1,70 @@ +# Task 018: Observability, Retry, Cancel, And Audit Trail + +Development description: Build the operational layer for workflow history, job logs, redaction, retry/cancel actions, failure UI, and audit events across the full pipeline. + +## Implementation Details + +- Workflow history records: + - Article created. + - Boundary questions generated. + - Boundary answers submitted. + - Plan generated/edited/approved. + - Research started. + - Evidence added/approved/rejected. + - Draft assembled. + - SEO/language review completed. + - Asset approved/rejected/replaced. + - Final approval granted. + - Publishing dry run completed. + - Publish commit created. + - Delayed publish verification failed. + - Site publishing config/script changed. + - Job failed/retried/cancelled. +- Failure UI shows: + - Job type. + - Status. + - Error category. + - Error message. + - Last successful step. + - Retry button when allowed. + - Admin logs. +- Sensitive values are redacted from logs before display. +- Retry rules: + - Plan generation manually only. + - Research allowed. + - Section scaffold per section. + - SEO/language review allowed. + - Publish commit allowed only if no publish commit exists. + +## Public Interface + +- Admin and Editor see workflow timeline. +- Admin can inspect redacted logs. +- Allowed retry/cancel actions are available in UI. + +## Acceptance Criteria + +- [ ] TDD pre-requirement: before implementation, write one failing behavior test showing a failed job appears in article workflow history with retry eligibility; proceed one operational behavior at a time and record evidence in `Result`. +- [ ] Article detail page shows user, system, and agent events in order. +- [ ] Job logs are stored and displayed with sensitive values redacted. +- [ ] Retry buttons appear only where retry is allowed. +- [ ] Cancelling queued/running jobs prevents article state mutation. +- [ ] Publish commit retry is blocked once a publish commit exists. +- [ ] Admin can see detailed logs; Editor sees safe failure summary. +- [ ] All Admin script/config changes are audit-visible with diff and rollback target. + +## Verification + +- Run audit/workflow API tests. +- Run log redaction tests. +- Run frontend timeline/failure UI tests. +- Run smoke flow that forces a failed fake job and retries it. + +## Result + +- Status: Pending execution. +- TDD plan: To be filled during execution. +- Red evidence: To be filled during execution. +- Green evidence: To be filled during execution. +- Refactor notes: To be filled during execution. +- Verification output: To be filled during execution. diff --git a/tasks/019-end-to-end-demo-stack.md b/tasks/019-end-to-end-demo-stack.md new file mode 100644 index 0000000..acd1fe3 --- /dev/null +++ b/tasks/019-end-to-end-demo-stack.md @@ -0,0 +1,64 @@ +# Task 019: End-To-End Demo Stack + +Development description: Assemble and verify a complete demo-ready Docker Compose product path from article brief to Git-backed publish commit using deterministic runner/research/review fixtures and a local demo Next content repository. + +## Implementation Details + +- Add demo fixtures: + - Seed target site. + - Demo Admin and Editor users. + - Fake Codex runner outputs for questions, plan, sections, SEO, language, assets. + - Fake research search/fetch corpus. + - Local bare Git repository or mounted demo Next content repository as publish target. +- Add end-to-end test path: + - Start stack. + - Create article. + - Generate/submit boundary answers. + - Generate/approve plan. + - Run research. + - Verify evidence matrix. + - Run production jobs. + - Assemble draft. + - Review assets, SEO, and language. + - Final approve. + - Dry run. + - Create publish commit. +- Add demo documentation: + - Setup. + - Login/role selection. + - Happy path. + - Failure demo path. + - How to inspect object storage, database state, and Git commit. + +## Public Interface + +- A reviewer can run one command, open the frontend, and complete the pipeline without external credentials. +- The final demo produces a real Git commit in the configured demo repository. + +## Acceptance Criteria + +- [ ] TDD pre-requirement: before implementation, write one failing end-to-end smoke test for the shortest happy path through public UI/API boundaries; add failure-path checks one behavior at a time and record evidence in `Result`. +- [ ] `docker compose up --build` starts the complete demo stack. +- [ ] Demo requires no real Codex, search, WHOIS, cloud S3, or GitHub credentials. +- [ ] Editor can complete the happy path from article brief to `PUBLISH_COMMIT_CREATED`. +- [ ] Final commit contains Markdown/MDX, frontmatter, and referenced assets according to site config. +- [ ] Object storage contains research source-section artifacts and manifests. +- [ ] Workflow timeline shows all major actions. +- [ ] Demo includes at least one visible failure/retry scenario. +- [ ] README/demo guide is accurate and enough for a new developer to run the product. + +## Verification + +- Run the full end-to-end smoke suite. +- Run `docker compose up --build` from a clean checkout. +- Manually execute the documented demo path. +- Inspect final Git commit, object storage artifacts, and workflow history. + +## Result + +- Status: Pending execution. +- TDD plan: To be filled during execution. +- Red evidence: To be filled during execution. +- Green evidence: To be filled during execution. +- Refactor notes: To be filled during execution. +- Verification output: To be filled during execution. diff --git a/tasks/README.md b/tasks/README.md new file mode 100644 index 0000000..4eeaee8 --- /dev/null +++ b/tasks/README.md @@ -0,0 +1,46 @@ +# AI Content Pipeline Task Pool + +This task pool turns `IDEA.md` into vertical, testable implementation slices that lead to a demo-ready Docker Compose product. + +## Execution Rules + +- Execute tasks in numeric order unless a later task is explicitly split or reprioritized. +- Every task starts with a development description and contains concrete implementation details. +- Every task has a TDD pre-requirement in acceptance criteria. +- Use vertical red-green-refactor cycles: one public behavior test, minimal implementation, then repeat. +- Do not write all tests first for a whole task. +- Fill each task's `Result` section after execution with: + - TDD plan. + - Red evidence. + - Green evidence. + - Refactor notes. + - Verification output. + +## Global Implementation Assumptions + +- v1 is internal or single-tenant. +- Roles are only `ADMIN` and `EDITOR`. +- Backend is FastAPI. +- Frontend is Next.js. +- Workflow orchestration uses LangGraph, while article/domain tables remain authoritative for product state. +- Queue is Redis-backed Celery, Dramatiq, or RQ. +- Postgres stores domain state and manifests, not large source snapshots. +- Object storage stores research source-section artifacts, assets, and job artifacts. +- Runner uses one managed Codex identity per environment, with fake-runner mode required for CI and demo. +- Publishing target is a Git-backed Next.js content repository. +- Publishing commits directly to the configured production branch after final approval and generic Markdown/MDX dry-run validation. +- Target repository CI/CD owns deployment after commit. +- Publishing validation is best-effort content-shape validation only; target-site build/runtime errors may reach production. +- Admin-defined publishing scripts run directly on the runner host inside checked-out site repositories; Admins are trusted code operators. + +## Demo Definition + +The pool is complete when a reviewer can: + +1. Run `docker compose up --build` from a clean checkout. +2. Open the frontend. +3. Use a demo Editor to create an article. +4. Complete boundary questions, plan approval, research, evidence review, production, draft review, final approval, dry run, and publish commit. +5. Inspect the final Git commit in the configured demo repository. +6. Inspect research artifacts in object storage. +7. Inspect workflow history and job logs in the UI. diff --git a/tests/smoke/public-health.sh b/tests/smoke/public-health.sh new file mode 100644 index 0000000..beba791 --- /dev/null +++ b/tests/smoke/public-health.sh @@ -0,0 +1,131 @@ +#!/usr/bin/env bash +set -euo pipefail + +ROOT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" +COMPOSE_FILE="${ROOT_DIR}/docker-compose.yml" + +BACKEND_URL="${BACKEND_URL:-http://localhost:8000}" +RUNNER_URL="${RUNNER_URL:-http://localhost:8010}" +FRONTEND_URL="${FRONTEND_URL:-http://localhost:3000}" +WAIT_SECONDS="${WAIT_SECONDS:-60}" +PYTHON_BIN="${PYTHON_BIN:-python3}" + +cleanup() { + docker compose -f "${COMPOSE_FILE}" down --remove-orphans >/dev/null 2>&1 || true +} + +fetch_health() { + local name="$1" + local url="$2" + local deadline=$((SECONDS + WAIT_SECONDS)) + local body + + while (( SECONDS < deadline )); do + if body="$(curl -fsS "${url}" 2>/dev/null)"; then + printf '%s' "${body}" + return 0 + fi + + sleep 1 + done + + echo "Timed out waiting for ${name} at ${url}" >&2 + return 1 +} + +assert_service_health() { + local name="$1" + local url="$2" + local expected_service="$3" + local body + + body="$(fetch_health "${name}" "${url}")" + HEALTH_NAME="${name}" EXPECTED_SERVICE="${expected_service}" HEALTH_BODY="${body}" "${PYTHON_BIN}" - <<'PY' +import json +import os +import sys + +name = os.environ["HEALTH_NAME"] +expected_service = os.environ["EXPECTED_SERVICE"] +body = os.environ["HEALTH_BODY"] + +try: + payload = json.loads(body) +except json.JSONDecodeError as error: + print(f"{name}: expected JSON response, got: {body!r}", file=sys.stderr) + raise SystemExit(1) from error + +if payload.get("service") != expected_service: + print( + f"{name}: expected service={expected_service!r}, got {payload.get('service')!r}", + file=sys.stderr, + ) + raise SystemExit(1) + +if payload.get("status") != "ok": + print(f"{name}: expected status='ok', got {payload.get('status')!r}", file=sys.stderr) + raise SystemExit(1) +PY + echo "${name} health ok" +} + +assert_backend_dependencies() { + local body + + body="$(fetch_health "backend dependencies" "${BACKEND_URL}/health/dependencies")" + HEALTH_BODY="${body}" "${PYTHON_BIN}" - <<'PY' +import json +import os +import sys + +body = os.environ["HEALTH_BODY"] +required = ("postgres", "redis", "object_storage") + +try: + payload = json.loads(body) +except json.JSONDecodeError as error: + print(f"backend dependencies: expected JSON response, got: {body!r}", file=sys.stderr) + raise SystemExit(1) from error + +if payload.get("service") != "backend": + print( + f"backend dependencies: expected service='backend', got {payload.get('service')!r}", + file=sys.stderr, + ) + raise SystemExit(1) + +if payload.get("status") != "ok": + print( + f"backend dependencies: expected status='ok', got {payload.get('status')!r}", + file=sys.stderr, + ) + raise SystemExit(1) + +dependencies = payload.get("dependencies") +if not isinstance(dependencies, dict): + print("backend dependencies: expected dependencies object", file=sys.stderr) + raise SystemExit(1) + +failed = [name for name in required if dependencies.get(name) != "ok"] +if failed: + print( + "backend dependencies: expected ok for " + ", ".join(failed), + file=sys.stderr, + ) + raise SystemExit(1) +PY + echo "backend dependencies ok" +} + +cd "${ROOT_DIR}" +trap cleanup EXIT + +echo "Starting stack with docker compose..." +docker compose -f "${COMPOSE_FILE}" up --build -d + +assert_service_health "backend" "${BACKEND_URL}/health" "backend" +assert_service_health "runner" "${RUNNER_URL}/health" "runner" +assert_service_health "frontend" "${FRONTEND_URL}/health" "frontend" +assert_backend_dependencies + +echo "public health smoke ok"