adding purpose-classifier data

This commit is contained in:
2026-07-29 23:45:26 -07:00
parent a0f34b89cb
commit c9d21bfc43
61 changed files with 9839 additions and 0 deletions
+127
View File
@@ -0,0 +1,127 @@
{"prompt": "type inference for our config language — hindley-milner-ish, unify across included files, and error messages that point at the right span", "purpose": "backendImpl", "secondary": null, "mixed": false, "difficulty": 0.9, "slice": "core", "lang": "en"}
{"prompt": "so the interpreter. tree-walking is fine for our sizes but the startup parse of a 40k line config is 900ms and people notice. bytecode vm, or caching the parsed ast, or a faster parser. i want the plan with what each buys and what it costs us in maintenance", "purpose": "planning", "secondary": null, "mixed": false, "difficulty": 0.9, "slice": "core", "lang": "en"}
{"prompt": "error messages say 'unexpected token' with no line number", "purpose": "quickFix", "secondary": null, "mixed": false, "difficulty": 0.3, "slice": "core", "lang": "en"}
{"prompt": "our ast nodes are boxed enums with Vec children everywhere. move to an arena with indices, identical evaluation results", "purpose": "refactor", "secondary": null, "mixed": false, "difficulty": 0.9, "slice": "core", "lang": "en"}
{"prompt": "stack overflow on deeply nested expressions, around 900 levels, and it takes the process down instead of erroring", "purpose": "debugging", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "core", "lang": "en"}
{"prompt": "explain the evaluation order in our interpreter for a chain of lazy references — i can't tell from reading it whether cycles are detected before or during evaluation", "purpose": "review", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "core", "lang": "en"}
{"prompt": "language reference for our config format: syntax, the type system, the standard functions, and the evaluation semantics", "purpose": "writing", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "boundary", "lang": "en"}
{"prompt": "formatter for the config language, idempotent, preserving comments and blank line groupings", "purpose": "backendImpl", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "core", "lang": "en"}
{"prompt": "the formatter drops trailing comments on the last line of a block", "purpose": "debugging", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "core", "lang": "en"}
{"prompt": "playground page for the config language: editor on the left, evaluated output on the right, live, with shareable urls", "purpose": "frontendImpl", "secondary": null, "mixed": false, "difficulty": 0.7, "slice": "core", "lang": "en"}
{"prompt": "the playground url gets too long for big configs and breaks in slack", "purpose": "quickFix", "secondary": null, "mixed": false, "difficulty": 0.3, "slice": "core", "lang": "en"}
{"prompt": "nuxt: the cms preview mode needs to render draft content without caching, and exit cleanly back to the published version", "purpose": "frontendImpl", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "core", "lang": "en"}
{"prompt": "preview mode leaks into the public site for the next visitor after an editor uses it", "purpose": "debugging", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "core", "lang": "en"}
{"prompt": "the preview cookie has no SameSite and no expiry", "purpose": "quickFix", "secondary": null, "mixed": false, "difficulty": 0.3, "slice": "core", "lang": "en"}
{"prompt": "our content fetching happens in asyncData in some pages and in a composable in others, with different error handling. one approach, same rendered output", "purpose": "refactor", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "core", "lang": "en"}
{"prompt": "plan the incremental static regeneration story — which pages, the revalidation triggers from the cms, and what an editor sees while it rebuilds", "purpose": "planning", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "core", "lang": "en"}
{"prompt": "webhook receiver that revalidates the affected pages when content changes, with a fallback full rebuild nightly", "purpose": "backendImpl", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "core", "lang": "en"}
{"prompt": "check whether an editor can publish content that breaks the build, and what happens to the live site if they do", "purpose": "review", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "core", "lang": "en"}
{"prompt": "editor handbook for the cms: content types, what each field does, the publishing flow, and the preview gotchas", "purpose": "writing", "secondary": null, "mixed": false, "difficulty": 0.5, "slice": "core", "lang": "en"}
{"prompt": "rich text renderer for the cms content with our components mapped to the block types, and a fallback for unknown blocks", "purpose": "frontendImpl", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "core", "lang": "en"}
{"prompt": "kafka: consumer group rebalances every few minutes and throughput tanks. session timeout is default, processing takes 2-3s per message", "purpose": "debugging", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "core", "lang": "en"}
{"prompt": "the max.poll.records is 500 with a 3 second per-message handler", "purpose": "quickFix", "secondary": null, "mixed": false, "difficulty": 0.3, "slice": "boundary", "lang": "en"}
{"prompt": "topic layout plan for the new event system — how many topics, partitioning key per topic, retention, and how we handle a topic that needs repartitioning later", "purpose": "planning", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "core", "lang": "en"}
{"prompt": "walk me through our consumer's offset commit strategy and tell me exactly when a message can be processed twice", "purpose": "review", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "core", "lang": "en"}
{"prompt": "onwards then", "purpose": "backendImpl", "secondary": null, "mixed": false, "difficulty": 0.3, "slice": "vague-eval", "lang": "en"}
{"prompt": "right so the queue situation. we have rabbitmq for the old services, kafka for the new event stream, and sqs for two lambdas. three sets of operational knowledge, three failure modes, three dashboards. i want a plan to get to one, or a defensible argument for keeping two. include the migration cost per service", "purpose": "planning", "secondary": null, "mixed": false, "difficulty": 0.9, "slice": "core", "lang": "en"}
{"prompt": "kafka consumer in the notification service replacing the rabbit consumer, same at-least-once semantics and the same dedupe behavior", "purpose": "backendImpl", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "core", "lang": "en"}
{"prompt": "queue metrics page: lag per consumer group, throughput, and the dlq counts, all in one view instead of three dashboards", "purpose": "frontendImpl", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "core", "lang": "en"}
{"prompt": "the rabbit prefetch is unlimited on one consumer", "purpose": "quickFix", "secondary": null, "mixed": false, "difficulty": 0.3, "slice": "core", "lang": "en"}
{"prompt": "our four consumers each implement their own retry-with-backoff-and-dlq. one library, same retry counts and dlq routing per consumer", "purpose": "refactor", "secondary": null, "mixed": false, "difficulty": 0.7, "slice": "core", "lang": "en"}
{"prompt": "messages land in the dlq with no error recorded, so we have no idea why they failed", "purpose": "debugging", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "core", "lang": "en"}
{"prompt": "explain what happens to an in-flight rabbit message when the consumer pod is SIGTERMed", "purpose": "review", "secondary": null, "mixed": false, "difficulty": 0.7, "slice": "core", "lang": "en"}
{"prompt": "runbook for a growing dlq: how to inspect a message, how to replay a batch, and when not to", "purpose": "writing", "secondary": null, "mixed": false, "difficulty": 0.5, "slice": "core", "lang": "en"}
{"prompt": "dlq replay tool with a filter, a dry run showing what would be replayed, and a rate limit so we don't self-ddos", "purpose": "backendImpl", "secondary": null, "mixed": false, "difficulty": 0.7, "slice": "core", "lang": "en"}
{"prompt": "dlq browser: paginated messages, the failure reason, the payload, and a replay-selected action", "purpose": "frontendImpl", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "core", "lang": "en"}
{"prompt": "replay sends the messages back to the front of the queue instead of the back", "purpose": "quickFix", "secondary": null, "mixed": false, "difficulty": 0.3, "slice": "core", "lang": "en"}
{"prompt": "plan the schema registry rollout — avro or protobuf, compatibility mode per topic, and how a producer gets blocked from breaking consumers", "purpose": "planning", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "core", "lang": "en"}
{"prompt": "a producer deployed a schema change and consumers started failing to deserialize, even though compatibility is supposedly enforced", "purpose": "debugging", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "core", "lang": "en"}
{"prompt": "tell me which of our topics have compatibility checking enabled and which are wide open", "purpose": "review", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "core", "lang": "en"}
{"prompt": "event schema documentation generated from the registry, one page per topic with the fields and their history", "purpose": "writing", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "boundary", "lang": "en"}
{"prompt": "producer library wrapper that registers schemas, validates before send, and fails loudly on an incompatible change at build time", "purpose": "backendImpl", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "core", "lang": "en"}
{"prompt": "our event payloads have both `userId` and `user_id` depending on the topic. normalize to snake_case with a compatibility window, and list every consumer that has to change", "purpose": "refactor", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "boundary", "lang": "en"}
{"prompt": "the schema registry url is hardcoded to the staging instance in one service", "purpose": "quickFix", "secondary": null, "mixed": false, "difficulty": 0.2, "slice": "core", "lang": "en"}
{"prompt": "plan the queue consolidation then implement the first consumer migration", "purpose": "planning", "secondary": "backendImpl", "mixed": true, "difficulty": 0.8, "slice": "mixed", "lang": "en"}
{"prompt": "figure out why the dlq has no error context and then write the runbook entry", "purpose": "debugging", "secondary": "writing", "mixed": true, "difficulty": 0.6, "slice": "mixed", "lang": "en"}
{"prompt": "carry on then", "purpose": "backendImpl", "secondary": null, "mixed": false, "difficulty": 0.3, "slice": "vague-eval", "lang": "en"}
{"prompt": "here's the failing build, the config language tests are red after i touched the lexer:\n\nrunning 214 tests\ntest lexer::tests::string_escapes ... FAILED\ntest lexer::tests::nested_interpolation ... FAILED\ntest parser::tests::multiline_string ... FAILED\n\nfailures:\n\n---- lexer::tests::string_escapes stdout ----\nthread 'lexer::tests::string_escapes' panicked at src/lexer.rs:412:9:\nassertion `left == right` failed\n left: [Str(\"a\\\\nb\"), Eof]\n right: [Str(\"a\\nb\"), Eof]\n\n---- lexer::tests::nested_interpolation stdout ----\nthread 'lexer::tests::nested_interpolation' panicked at src/lexer.rs:455:9:\nassertion `left == right` failed\n left: [StrStart, Lit(\"x=\"), InterpStart, Ident(\"a\"), InterpEnd, StrEnd, Eof]\n right: [StrStart, Lit(\"x=\"), InterpStart, Ident(\"a\"), Plus, Ident(\"b\"), InterpEnd, StrEnd, Eof]\n\ntest result: FAILED. 211 passed; 3 failed\n\ni was trying to handle raw strings and clearly broke escape processing and interpolation", "purpose": "debugging", "secondary": null, "mixed": false, "difficulty": 0.7, "slice": "pasted-context", "lang": "en"}
{"prompt": "raw string literals in the lexer, r\"...\" style, without touching how escapes work in normal strings", "purpose": "backendImpl", "secondary": null, "mixed": false, "difficulty": 0.7, "slice": "core", "lang": "en"}
{"prompt": "our lexer's string handling is one 300 line function with nested state flags. split it into explicit states, same token stream for the whole test corpus", "purpose": "refactor", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "core", "lang": "en"}
{"prompt": "string interpolation nested two deep produces the wrong token order", "purpose": "debugging", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "core", "lang": "en"}
{"prompt": "syntax highlighting grammar for our config language, for vscode and for the docs site's code blocks", "purpose": "frontendImpl", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "core", "lang": "en"}
{"prompt": "the tmLanguage file doesn't highlight interpolation inside strings", "purpose": "quickFix", "secondary": null, "mixed": false, "difficulty": 0.3, "slice": "core", "lang": "en"}
{"prompt": "review our lexer's handling of unicode identifiers and tell me whether we're consistent with what the docs claim", "purpose": "review", "secondary": null, "mixed": false, "difficulty": 0.7, "slice": "boundary", "lang": "en"}
{"prompt": "spec section on string literals: the escape table, raw strings, multiline behavior, and interpolation rules", "purpose": "writing", "secondary": null, "mixed": false, "difficulty": 0.7, "slice": "boundary", "lang": "en"}
{"prompt": "plan how we version the config language itself so we can make a breaking syntax change without breaking every existing file", "purpose": "planning", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "core", "lang": "en"}
{"prompt": "fuzzing setup for the parser with a corpus from our customers' configs, running in CI nightly", "purpose": "backendImpl", "secondary": null, "mixed": false, "difficulty": 0.7, "slice": "core", "lang": "en"}
{"prompt": "the fuzzer found a panic on an unterminated block comment", "purpose": "quickFix", "secondary": null, "mixed": false, "difficulty": 0.4, "slice": "core", "lang": "en"}
{"prompt": "explain how our error recovery works when parsing an invalid file — do we report one error or try to continue", "purpose": "review", "secondary": null, "mixed": false, "difficulty": 0.7, "slice": "core", "lang": "en"}
{"prompt": "diagnostics rendering with the source snippet, a caret span, and a note when there's a likely fix — like rustc's output", "purpose": "backendImpl", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "core", "lang": "en"}
{"prompt": "error output uses ansi colors even when piped to a file", "purpose": "quickFix", "secondary": null, "mixed": false, "difficulty": 0.2, "slice": "core", "lang": "en"}
{"prompt": "error message catalog doc — every diagnostic code, what causes it, and how to fix it", "purpose": "writing", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "boundary", "lang": "en"}
{"prompt": "map out the plan for making the interpreter embeddable — a c api, a stable abi, and what we promise about memory ownership", "purpose": "planning", "secondary": null, "mixed": false, "difficulty": 0.9, "slice": "core", "lang": "en"}
{"prompt": "wieder mal: die Cache-Invalidierung im CMS-Frontend greift nicht, wenn nur ein verschachtelter Block geändert wird. woran liegt das", "purpose": "debugging", "secondary": null, "mixed": false, "difficulty": 0.7, "slice": "core", "lang": "de"}
{"prompt": "content model for a page builder: sections with typed props, ordering, and per-section visibility rules", "purpose": "backendImpl", "secondary": null, "mixed": false, "difficulty": 0.7, "slice": "core", "lang": "en"}
{"prompt": "page builder ui: section list with drag reorder, an inline settings panel per section, and a live preview pane", "purpose": "frontendImpl", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "core", "lang": "en"}
{"prompt": "the preview pane reloads the whole iframe on every keystroke", "purpose": "quickFix", "secondary": null, "mixed": false, "difficulty": 0.4, "slice": "boundary", "lang": "en"}
{"prompt": "we have two component registries, one for the renderer and one for the editor's settings forms, and adding a component means touching both. unify, same components available in both", "purpose": "refactor", "secondary": null, "mixed": false, "difficulty": 0.7, "slice": "core", "lang": "en"}
{"prompt": "sections reorder correctly in the editor but publish in the original order", "purpose": "debugging", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "core", "lang": "en"}
{"prompt": "check whether an editor can inject html through any of the rich text fields and get it rendered unescaped", "purpose": "review", "secondary": null, "mixed": false, "difficulty": 0.7, "slice": "core", "lang": "en"}
{"prompt": "documentation for adding a new page-builder section, aimed at a developer joining next week", "purpose": "writing", "secondary": null, "mixed": false, "difficulty": 0.5, "slice": "core", "lang": "en"}
{"prompt": "plan the localization of cms content — per-locale drafts, fallbacks, and what an editor sees when a translation is missing", "purpose": "planning", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "core", "lang": "en"}
{"prompt": "ok so", "purpose": "planning", "secondary": null, "mixed": false, "difficulty": 0.3, "slice": "vague-eval", "lang": "en"}
{"prompt": "i want the design for our multi-region active-active setup. writes in both regions, conflict resolution, and a story for the tables where conflicts are unacceptable (billing, primarily). also what we tell customers about consistency, because right now we'd be lying if we said strong", "purpose": "planning", "secondary": null, "mixed": false, "difficulty": 0.9, "slice": "core", "lang": "en"}
{"prompt": "region-aware routing in the api so a write goes to the home region of the tenant, with a redirect rather than a cross-region write", "purpose": "backendImpl", "secondary": null, "mixed": false, "difficulty": 0.9, "slice": "core", "lang": "en"}
{"prompt": "region indicator in the app header for internal builds, showing which region served the request", "purpose": "frontendImpl", "secondary": null, "mixed": false, "difficulty": 0.4, "slice": "core", "lang": "en"}
{"prompt": "the region config is read from an env var that isn't set in one deployment, defaulting to us-east", "purpose": "quickFix", "secondary": null, "mixed": false, "difficulty": 0.3, "slice": "core", "lang": "en"}
{"prompt": "our db access assumes a single connection string throughout. thread the region through properly, with the same queries hitting the same data as today", "purpose": "refactor", "secondary": null, "mixed": false, "difficulty": 0.9, "slice": "core", "lang": "en"}
{"prompt": "requests from europe occasionally get served by the us region and take 400ms extra, but only for authenticated users", "purpose": "debugging", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "core", "lang": "en"}
{"prompt": "explain what consistency our current setup actually provides for a read immediately after a write from a different region", "purpose": "review", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "core", "lang": "en"}
{"prompt": "customer-facing doc on data residency and regions: where data lives, which operations are cross-region, and the latency implications", "purpose": "writing", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "boundary", "lang": "en"}
{"prompt": "tenant region migration: move a tenant's data to another region with a short write freeze and a verifiable cutover", "purpose": "backendImpl", "secondary": null, "mixed": false, "difficulty": 0.9, "slice": "core", "lang": "en"}
{"prompt": "region migration admin ui: pick a tenant, pick a target, show the estimated freeze, and a progress log", "purpose": "frontendImpl", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "core", "lang": "en"}
{"prompt": "the migration progress log shows percentages over 100", "purpose": "quickFix", "secondary": null, "mixed": false, "difficulty": 0.2, "slice": "core", "lang": "en"}
{"prompt": "plan how we test region failover for real, including the customer comms and the rollback", "purpose": "planning", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "core", "lang": "en"}
{"prompt": "after a failover drill, some tenants had their region flag pointing at the old region for hours", "purpose": "debugging", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "core", "lang": "en"}
{"prompt": "read the failover automation and tell me which steps are actually automated versus documented", "purpose": "review", "secondary": null, "mixed": false, "difficulty": 0.7, "slice": "boundary", "lang": "en"}
{"prompt": "failover runbook, honest about the manual steps, with the exact commands and the verification at each stage", "purpose": "writing", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "core", "lang": "en"}
{"prompt": "our health checks report per-pod health but nothing reports regional health. add a regional readiness signal the dns layer can use", "purpose": "backendImpl", "secondary": null, "mixed": false, "difficulty": 0.7, "slice": "core", "lang": "en"}
{"prompt": "next bit then", "purpose": "backendImpl", "secondary": null, "mixed": false, "difficulty": 0.3, "slice": "vague-eval", "lang": "en"}
{"prompt": "here's the ticket that's been bouncing around for a month, nobody wants it:\n\nTITLE: Bulk tag update times out for large accounts\nDETAIL: PATCH /v1/records/bulk-tag with 5000 record ids returns 504 after 60s for accounts\n with >2M records. Works fine under 500 ids. The gateway timeout is 60s and can't be raised.\nWHAT I TRIED: adding an index on record_tags(record_id) — no change. The slow part appears\n to be the tag reconciliation, we delete all tags for each record then reinsert.\nCUSTOMER: three enterprise accounts hitting this weekly, they've resorted to 200-at-a-time\n loops which take 40 minutes.\nAPPETITE: whatever it takes, this is embarrassing", "purpose": "backendImpl", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "pasted-context", "lang": "en"}
{"prompt": "make bulk tagging async — accept the request, return a job id, and provide a status endpoint with per-record results", "purpose": "backendImpl", "secondary": null, "mixed": false, "difficulty": 0.7, "slice": "core", "lang": "en"}
{"prompt": "the delete-then-insert tag reconciliation should be a diff — compute added and removed, touch only those rows", "purpose": "refactor", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "boundary", "lang": "en"}
{"prompt": "bulk action ui that submits async and shows a progress toast with a link to the job result", "purpose": "frontendImpl", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "core", "lang": "en"}
{"prompt": "the bulk endpoint accepts unlimited ids, cap it at 10000 with a clear error", "purpose": "quickFix", "secondary": null, "mixed": false, "difficulty": 0.3, "slice": "core", "lang": "en"}
{"prompt": "review our other bulk endpoints for the same delete-then-insert pattern", "purpose": "review", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "core", "lang": "en"}
{"prompt": "api docs for the async bulk pattern — submit, poll, interpret the per-item results, and the limits", "purpose": "writing", "secondary": null, "mixed": false, "difficulty": 0.5, "slice": "boundary", "lang": "en"}
{"prompt": "plan the async job pattern properly so every long operation uses it instead of each endpoint inventing its own", "purpose": "planning", "secondary": null, "mixed": false, "difficulty": 0.7, "slice": "core", "lang": "en"}
{"prompt": "job status polling hammers us — 4 requests per second per client because the ui polls at 250ms", "purpose": "debugging", "secondary": null, "mixed": false, "difficulty": 0.5, "slice": "core", "lang": "en"}
{"prompt": "job progress via server-sent events instead of polling, with a polling fallback", "purpose": "backendImpl", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "core", "lang": "en"}
{"prompt": "tag picker with create-on-type, existing tags fuzzy matched, and a limit indicator when the record has many", "purpose": "frontendImpl", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "core", "lang": "en"}
{"prompt": "creating a tag with a leading space creates a duplicate that looks identical", "purpose": "quickFix", "secondary": null, "mixed": false, "difficulty": 0.3, "slice": "core", "lang": "en"}
{"prompt": "our tag normalization happens in the ui for creation and in the api for import, differently, which is where the duplicates come from. one normalizer, and dedupe the existing ones", "purpose": "refactor", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "core", "lang": "en"}
{"prompt": "explain how tag permissions work — can a viewer add tags, and is that enforced server side", "purpose": "review", "secondary": null, "mixed": false, "difficulty": 0.5, "slice": "core", "lang": "en"}
{"prompt": "help article on tags: creating, renaming, merging, and what happens to records when you delete a tag", "purpose": "writing", "secondary": null, "mixed": false, "difficulty": 0.4, "slice": "core", "lang": "en"}
{"prompt": "tag merge operation that reassigns records and keeps an alias so old api calls with the merged tag still work", "purpose": "backendImpl", "secondary": null, "mixed": false, "difficulty": 0.7, "slice": "core", "lang": "en"}
{"prompt": "plan how we'd support hierarchical tags without breaking the flat api", "purpose": "planning", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "core", "lang": "en"}
{"prompt": "renaming a tag to an existing name silently merges them with no warning", "purpose": "debugging", "secondary": null, "mixed": false, "difficulty": 0.5, "slice": "boundary", "lang": "en"}
{"prompt": "ya, dale", "purpose": "backendImpl", "secondary": null, "mixed": false, "difficulty": 0.3, "slice": "vague-eval", "lang": "es"}
{"prompt": "propose the architecture for our ai features. we want summarization of a record's activity, a natural language query over the user's data, and suggested tags. constraints: no customer data leaves our vpc without an opt-in, latency under 3s for the interactive ones, and cost per tenant has to be attributable", "purpose": "planning", "secondary": null, "mixed": false, "difficulty": 0.9, "slice": "core", "lang": "en"}
{"prompt": "summarization endpoint that assembles the context, calls the model, caches by content hash, and records token usage per tenant", "purpose": "backendImpl", "secondary": null, "mixed": false, "difficulty": 0.7, "slice": "core", "lang": "en"}
{"prompt": "summary panel with a streaming render, a regenerate action, and a clear 'ai generated' marker", "purpose": "frontendImpl", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "core", "lang": "en"}
{"prompt": "the streaming render flickers because we re-render the whole markdown on every chunk", "purpose": "quickFix", "secondary": null, "mixed": false, "difficulty": 0.4, "slice": "boundary", "lang": "en"}
{"prompt": "our three ai call sites each build prompts inline and handle errors differently. one client with the prompts in a registry, same prompts sent", "purpose": "refactor", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "core", "lang": "en"}
{"prompt": "summaries occasionally include content from a different record, which is the worst possible bug here", "purpose": "debugging", "secondary": null, "mixed": false, "difficulty": 0.9, "slice": "core", "lang": "en"}
{"prompt": "review the context assembly for the summarizer and tell me whether it can ever include data the requesting user can't see", "purpose": "review", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "core", "lang": "en"}
{"prompt": "customer-facing doc about our ai features: what data is used, where it goes, retention, and how to turn it off", "purpose": "writing", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "core", "lang": "en"}
{"prompt": "natural language to query translation with a validation pass that rejects anything touching another tenant, and shows the generated query to the user", "purpose": "backendImpl", "secondary": null, "mixed": false, "difficulty": 0.9, "slice": "core", "lang": "en"}
{"prompt": "query assistant ui: input, the generated query shown for review, run, and results with a 'this might be wrong' framing", "purpose": "frontendImpl", "secondary": null, "mixed": false, "difficulty": 0.7, "slice": "core", "lang": "en"}
{"prompt": "the token limit truncates long records silently mid-sentence", "purpose": "quickFix", "secondary": null, "mixed": false, "difficulty": 0.3, "slice": "core", "lang": "en"}
{"prompt": "plan the evaluation harness for these features — a golden set, what we measure, and how we catch a regression before customers do", "purpose": "planning", "secondary": null, "mixed": false, "difficulty": 0.8, "slice": "core", "lang": "en"}
{"prompt": "suggested tags are heavily biased toward the four most common tags regardless of content", "purpose": "debugging", "secondary": null, "mixed": false, "difficulty": 0.7, "slice": "core", "lang": "en"}
{"prompt": "check how we attribute model cost per tenant, and whether a shared cached response gets billed to the wrong one", "purpose": "review", "secondary": null, "mixed": false, "difficulty": 0.7, "slice": "core", "lang": "en"}
{"prompt": "internal doc on our prompt registry: how to add a prompt, the versioning, and the review requirement before it ships", "purpose": "writing", "secondary": null, "mixed": false, "difficulty": 0.5, "slice": "core", "lang": "en"}
{"prompt": "per-tenant ai usage limits with a clear message when exceeded and an admin view of consumption", "purpose": "backendImpl", "secondary": null, "mixed": false, "difficulty": 0.6, "slice": "core", "lang": "en"}
{"prompt": "design the eval harness and then build the golden set runner", "purpose": "planning", "secondary": "backendImpl", "mixed": true, "difficulty": 0.8, "slice": "mixed", "lang": "en"}
{"prompt": "explain how the context assembly works then write it up for the security review", "purpose": "review", "secondary": "writing", "mixed": true, "difficulty": 0.7, "slice": "mixed", "lang": "en"}
{"prompt": "one more and then stop", "purpose": "quickFix", "secondary": null, "mixed": false, "difficulty": 0.3, "slice": "vague-eval", "lang": "en"}