Skip to main content
EVOKORE// BROWSE
>

./browse/prompts

13 NODES
πŸ€–system promptβ€’7 months ago

solidity-security

Master smart contract security best practices to prevent common

security
⭐1
# Solidity Security Master smart contract security best practices, vulnerability prevention, and secure Solidity development patterns. ## When to Use This Skill - Writing secure smart contracts - Auditing existing contracts for vulnerabilities - Implementing secure DeFi protocols - Preventing reentrancy, overflow, and access control issues - Optimizing gas usage while maintaining security - Preparing contracts for professional audits - Understanding common attack vectors ## Critical Vulnerabilities ### 1. Reentrancy Attacker calls back into your contract before state is updated. **Vulnerable Code:** ```solidity // VULNERABLE TO REENTRANCY contract VulnerableBank { mapping(address => uint256) public balances; function withdraw() public { uint256 amount = balances[msg.sender]; // DANGER: External call before state update (bool success, ) = msg.sender.call{value: amount}(""); require(success); balances[msg.sender] = 0; // Too late! } } ``` **Secure Pattern (Checks-Effects-Interactions):** ```solidity contract SecureBank { mapping(address => uint256) public balances; function withdraw() public { uint256 amount = balances[msg.sender]; require(amount > 0, "Insufficient balance"); // EFFECTS: Update state BEFORE external call balances[msg.sender] = 0; // INTERACTIONS: External call last (bool success, ) = msg.sender.call{value: amount}(""); require(success, "Transfer failed"); } } ``` **Alternative: ReentrancyGuard** ```solidity import "@openzeppelin/contracts/security/ReentrancyGuard.sol"; contract SecureBank is ReentrancyGuard { mapping(address => uint256) public balances; function withdraw() public nonReentrant { uint256 amount = balances[msg.sender]; require(amount > 0, "Insufficient balance"); balances[msg.sender] = 0; (bool success, ) = msg.sender.call{value: amount}(""); require(success, "Transfer failed"); } } ``` ### 2. Integer Overflow/Underflow **Vulnerable Code (Solidity < 0.8.0):** ```solidity // VULNERABLE contract VulnerableToken { mapping(address => uint256) public balances; function transfer(address to, uint256 amount) public { // No overflow check - can wrap around balances[msg.sender] -= amount; // Can underflow! balances[to] += amount; // Can overflow! } } ``` **Secure Pattern (Solidity >= 0.8.0):** ```solidity // Solidity 0.8+ has built-in overflow/underflow checks contract SecureToken { mapping(address => uint256) public balances; function transfer(address to, uint256 amount) public { // Automatically reverts on overflow/underflow balances[msg.sender] -= amount; balances[to] += amount; } } ``` **For Solidity < 0.8.0, use SafeMath:** ```solidity import "@openzeppelin/contracts/utils/math/SafeMath.sol"; contract SecureToken { using SafeMath for uint256; mapping(address => uint256) public balances; function transfer(address to, uint256 amount) public { balances[msg.sender] = balances[msg.sender].sub(amount); balances[to] = balances[to].add(amount); } } ``` ### 3. Access Control **Vulnerable Code:** ```solidity // VULNERABLE: Anyone can call critical functions contract VulnerableContract { address public owner; function withdraw(uint256 amount) public { // No access control! payable(msg.sender).transfer(amount); } } ``` **Secure Pattern:** ```solidity import "@openzeppelin/contracts/access/Ownable.sol"; contract SecureContract is Ownable { function withdraw(uint256 amount) public onlyOwner { payable(owner()).transfer(amount); } } // Or implement custom role-based access contract RoleBasedContract { mapping(address => bool) public admins; modifier onlyAdmin() { require(admins[msg.sender], "Not an admin"); _; } function criticalFunction() public onlyAdmin { // Protected function } } ``` ### 4. Front-Running **Vulnerable:** ```solidity // VULNERABLE TO FRONT-RUNNING contract VulnerableDEX { function swap(uint256 amount, uint256 minOutput) public { // Attacker sees this in mempool and front-runs uint256 output = calculateOutput(amount); require(output >= minOutput, "Slippage too high"); // Perform swap } } ``` **Mitigation:** ```solidity contract SecureDEX { mapping(bytes32 => bool) public usedCommitments; // Step 1: Commit to trade function commitTrade(bytes32 commitment) public { usedCommitments[commitment] = true; } // Step 2: Reveal trade (next block) function revealTrade( uint256 amount, uint256 minOutput, bytes32 secret ) public { bytes32 commitment = keccak256(abi.encodePacked( msg.sender, amount, minOutput, secret )); require(usedCommitments[commitment], "Invalid commitment"); // Perform swap } } ``` ## Security Best Practices ### Checks-Effects-Interactions Pattern ```solidity contract SecurePattern { mapping(address => uint256) public balances; function withdraw(uint256 amount) public { // 1. CHECKS: Validate conditions require(amount <= balances[msg.sender], "Insufficient balance"); require(amount > 0, "Amount must be positive"); // 2. EFFECTS: Update state balances[msg.sender] -= amount; // 3. INTERACTIONS: External calls last (bool success, ) = msg.sender.call{value: amount}(""); require(success, "Transfer failed"); } } ``` ### Pull Over Push Pattern ```solidity // Prefer this (pull) contract SecurePayment { mapping(address => uint256) public pendingWithdrawals; function recordPayment(address recipient, uint256 amount) internal { pendingWithdrawals[recipient] += amount; } function withdraw() public { uint256 amount = pendingWithdrawals[msg.sender]; require(amount > 0, "Nothing to withdraw"); pendingWithdrawals[msg.sender] = 0; payable(msg.sender).transfer(amount); } } // Over this (push) contract RiskyPayment { function distributePayments(address[] memory recipients, uint256[] memory amounts) public { for (uint i = 0; i < recipients.length; i++) { // If any transfer fails, entire batch fails payable(recipients[i]).transfer(amounts[i]); } } } ``` ### Input Validation ```solidity contract SecureContract { function transfer(address to, uint256 amount) public { // Validate inputs require(to != address(0), "Invalid recipient"); require(to != address(this), "Cannot send to contract"); require(amount > 0, "Amount must be positive"); require(amount <= balances[msg.sender], "Insufficient balance"); // Proceed with transfer balances[msg.sender] -= amount; balances[to] += amount; } } ``` ### Emergency Stop (Circuit Breaker) ```solidity import "@openzeppelin/contracts/security/Pausable.sol"; contract EmergencyStop is Pausable, Ownable { function criticalFunction() public whenNotPaused { // Function logic } function emergencyStop() public onlyOwner { _pause(); } function resume() public onlyOwner { _unpause(); } } ``` ## Gas Optimization ### Use `uint256` Instead of Smaller Types ```solidity // More gas efficient contract GasEfficient { uint256 public value; // Optimal function set(uint256 _value) public { value = _value; } } // Less efficient contract GasInefficient { uint8 public value; // Still uses 256-bit slot function set(uint8 _value) public { value = _value; // Extra gas for type conversion } } ``` ### Pack Storage Variables ```solidity // Gas efficient (3 variables in 1 slot) contract PackedStorage { uint128 public a; // Slot 0 uint64 public b; // Slot 0 uint64 public c; // Slot 0 uint256 public d; // Slot 1 } // Gas inefficient (each variable in separate slot) contract UnpackedStorage { uint256 public a; // Slot 0 uint256 public b; // Slot 1 uint256 public c; // Slot 2 uint256 public d; // Slot 3 } ``` ### Use `calldata` Instead of `memory` for Function Arguments ```solidity contract GasOptimized { // More gas efficient function processData(uint256[] calldata data) public pure returns (uint256) { return data[0]; } // Less efficient function processDataMemory(uint256[] memory data) public pure returns (uint256) { return data[0]; } } ``` ### Use Events for Data Storage (When Appropriate) ```solidity contract EventStorage { // Emitting events is cheaper than storage event DataStored(address indexed user, uint256 indexed id, bytes data); function storeData(uint256 id, bytes calldata data) public { emit DataStored(msg.sender, id, data); // Don't store in contract storage unless needed } } ``` ## Common Vulnerabilities Checklist ```solidity // Security Checklist Contract contract SecurityChecklist { /** * [ ] Reentrancy protection (ReentrancyGuard or CEI pattern) * [ ] Integer overflow/underflow (Solidity 0.8+ or SafeMath) * [ ] Access control (Ownable, roles, modifiers) * [ ] Input validation (require statements) * [ ] Front-running mitigation (commit-reveal if applicable) * [ ] Gas optimization (packed storage, calldata) * [ ] Emergency stop mechanism (Pausable) * [ ] Pull over push pattern for payments * [ ] No delegatecall to untrusted contracts * [ ] No tx.origin for authentication (use msg.sender) * [ ] Proper event emission * [ ] External calls at end of function * [ ] Check return values of external calls * [ ] No hardcoded addresses * [ ] Upgrade mechanism (if proxy pattern) */ } ``` ## Testing for Security ```javascript // Hardhat test example const { expect } = require("chai"); const { ethers } = require("hardhat"); describe("Security Tests", function () { it("Should prevent reentrancy attack", async function () { const [attacker] = await ethers.getSigners(); const VictimBank = await ethers.getContractFactory("SecureBank"); const bank = await VictimBank.deploy(); const Attacker = await ethers.getContractFactory("ReentrancyAttacker"); const attackerContract = await Attacker.deploy(bank.address); // Deposit funds await bank.deposit({ value: ethers.utils.parseEther("10") }); // Attempt reentrancy attack await expect( attackerContract.attack({ value: ethers.utils.parseEther("1") }), ).to.be.revertedWith("ReentrancyGuard: reentrant call"); }); it("Should prevent integer overflow", async function () { const Token = await ethers.getContractFactory("SecureToken"); const token = await Token.deploy(); // Attempt overflow await expect(token.transfer(attacker.address, ethers.constants.MaxUint256)) .to.be.reverted; }); it("Should enforce access control", async function () { const [owner, attacker] = await ethers.getSigners(); const Contract = await ethers.getContractFactory("SecureContract"); const contract = await Contract.deploy(); // Attempt unauthorized withdrawal await expect(contract.connect(attacker).withdraw(100)).to.be.revertedWith( "Ownable: caller is not the owner", ); }); }); ``` ## Audit Preparation ```solidity contract WellDocumentedContract { /** * @title Well Documented Contract * @dev Example of proper documentation for audits * @notice This contract handles user deposits and withdrawals */ /// @notice Mapping of user balances mapping(address => uint256) public balances; /** * @dev Deposits ETH into the contract * @notice Anyone can deposit funds */ function deposit() public payable { require(msg.value > 0, "Must send ETH"); balances[msg.sender] += msg.value; } /** * @dev Withdraws user's balance * @notice Follows CEI pattern to prevent reentrancy * @param amount Amount to withdraw in wei */ function withdraw(uint256 amount) public { // CHECKS require(amount <= balances[msg.sender], "Insufficient balance"); // EFFECTS balances[msg.sender] -= amount; // INTERACTIONS (bool success, ) = msg.sender.call{value: amount}(""); require(success, "Transfer failed"); } } ``` ## Resources - **references/reentrancy.md**: Comprehensive reentrancy prevention - **references/access-control.md**: Role-based access patterns - **references/overflow-underflow.md**: SafeMath and integer safety - **references/gas-optimization.md**: Gas saving techniques - **references/vulnerability-patterns.md**: Common vulnerability catalog - **assets/solidity-contracts-templates.sol**: Secure contract templates - **assets/security-checklist.md**: Pre-audit checklist - **scripts/analyze-contract.sh**: Static analysis tools ## Tools for Security Analysis - **Slither**: Static analysis tool - **Mythril**: Security analysis tool - **Echidna**: Fuzzing tool - **Manticore**: Symbolic execution - **Securify**: Automated security scanner ## Common Pitfalls 1. **Using `tx.origin` for Authentication**: Use `msg.sender` instead 2. **Unchecked External Calls**: Always check return values 3. **Delegatecall to Untrusted Contracts**: Can hijack your contract 4. **Floating Pragma**: Pin to specific Solidity version 5. **Missing Events**: Emit events for state changes 6. **Excessive Gas in Loops**: Can hit block gas limit 7. **No Upgrade Path**: Consider proxy patterns if upgrades needed
πŸ‘0
πŸ‘οΈ0
πŸ€– Auto-discovered
πŸ€–system promptβ€’7 months ago

data-storytelling

Transform data into compelling narratives using visualization,

data
⭐1
# Data Storytelling Transform raw data into compelling narratives that drive decisions and inspire action. ## When to Use This Skill - Presenting analytics to executives - Creating quarterly business reviews - Building investor presentations - Writing data-driven reports - Communicating insights to non-technical audiences - Making recommendations based on data ## Core Concepts ### 1. Story Structure ``` Setup β†’ Conflict β†’ Resolution Setup: Context and baseline Conflict: The problem or opportunity Resolution: Insights and recommendations ``` ### 2. Narrative Arc ``` 1. Hook: Grab attention with surprising insight 2. Context: Establish the baseline 3. Rising Action: Build through data points 4. Climax: The key insight 5. Resolution: Recommendations 6. Call to Action: Next steps ``` ### 3. Three Pillars | Pillar | Purpose | Components | | ------------- | -------- | -------------------------------- | | **Data** | Evidence | Numbers, trends, comparisons | | **Narrative** | Meaning | Context, causation, implications | | **Visuals** | Clarity | Charts, diagrams, highlights | ## Story Frameworks ### Framework 1: The Problem-Solution Story ```markdown # Customer Churn Analysis ## The Hook "We're losing $2.4M annually to preventable churn." ## The Context - Current churn rate: 8.5% (industry average: 5%) - Average customer lifetime value: $4,800 - 500 customers churned last quarter ## The Problem Analysis of churned customers reveals a pattern: - 73% churned within first 90 days - Common factor: < 3 support interactions - Low feature adoption in first month ## The Insight [Show engagement curve visualization] Customers who don't engage in the first 14 days are 4x more likely to churn. ## The Solution 1. Implement 14-day onboarding sequence 2. Proactive outreach at day 7 3. Feature adoption tracking ## Expected Impact - Reduce early churn by 40% - Save $960K annually - Payback period: 3 months ## Call to Action Approve $50K budget for onboarding automation. ``` ### Framework 2: The Trend Story ```markdown # Q4 Performance Analysis ## Where We Started Q3 ended with $1.2M MRR, 15% below target. Team morale was low after missed goals. ## What Changed [Timeline visualization] - Oct: Launched self-serve pricing - Nov: Reduced friction in signup - Dec: Added customer success calls ## The Transformation [Before/after comparison chart] | Metric | Q3 | Q4 | Change | |----------------|--------|--------|--------| | Trial β†’ Paid | 8% | 15% | +87% | | Time to Value | 14 days| 5 days | -64% | | Expansion Rate | 2% | 8% | +300% | ## Key Insight Self-serve + high-touch creates compound growth. Customers who self-serve AND get a success call have 3x higher expansion rate. ## Going Forward Double down on hybrid model. Target: $1.8M MRR by Q2. ``` ### Framework 3: The Comparison Story ```markdown # Market Opportunity Analysis ## The Question Should we expand into EMEA or APAC first? ## The Comparison [Side-by-side market analysis] ### EMEA - Market size: $4.2B - Growth rate: 8% - Competition: High - Regulatory: Complex (GDPR) - Language: Multiple ### APAC - Market size: $3.8B - Growth rate: 15% - Competition: Moderate - Regulatory: Varied - Language: Multiple ## The Analysis [Weighted scoring matrix visualization] | Factor | Weight | EMEA Score | APAC Score | | ----------- | ------ | ---------- | ---------- | | Market Size | 25% | 5 | 4 | | Growth | 30% | 3 | 5 | | Competition | 20% | 2 | 4 | | Ease | 25% | 2 | 3 | | **Total** | | **2.9** | **4.1** | ## The Recommendation APAC first. Higher growth, less competition. Start with Singapore hub (English, business-friendly). Enter EMEA in Year 2 with localization ready. ## Risk Mitigation - Timezone coverage: Hire 24/7 support - Cultural fit: Local partnerships - Payment: Multi-currency from day 1 ``` ## Visualization Techniques ### Technique 1: Progressive Reveal ```markdown Start simple, add layers: Slide 1: "Revenue is growing" [single line chart] Slide 2: "But growth is slowing" [add growth rate overlay] Slide 3: "Driven by one segment" [add segment breakdown] Slide 4: "Which is saturating" [add market share] Slide 5: "We need new segments" [add opportunity zones] ``` ### Technique 2: Contrast and Compare ```markdown Before/After: β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”¬β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” β”‚ BEFORE β”‚ AFTER β”‚ β”‚ β”‚ β”‚ β”‚ Process: 5 daysβ”‚ Process: 1 day β”‚ β”‚ Errors: 15% β”‚ Errors: 2% β”‚ β”‚ Cost: $50/unit β”‚ Cost: $20/unit β”‚ β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”΄β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜ This/That (emphasize difference): β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” β”‚ CUSTOMER A vs B β”‚ β”‚ β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” β”‚ β”‚ β”‚ β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ β”‚ β”‚ β–ˆβ–ˆ β”‚ β”‚ β”‚ β”‚ $45,000 β”‚ β”‚ $8,000 β”‚ β”‚ β”‚ β”‚ LTV β”‚ β”‚ LTV β”‚ β”‚ β”‚ β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜ β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜ β”‚ β”‚ Onboarded No onboarding β”‚ β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜ ``` ### Technique 3: Annotation and Highlight ```python import matplotlib.pyplot as plt import pandas as pd fig, ax = plt.subplots(figsize=(12, 6)) # Plot the main data ax.plot(dates, revenue, linewidth=2, color='#2E86AB') # Add annotation for key events ax.annotate( 'Product Launch\n+32% spike', xy=(launch_date, launch_revenue), xytext=(launch_date, launch_revenue * 1.2), fontsize=10, arrowprops=dict(arrowstyle='->', color='#E63946'), color='#E63946' ) # Highlight a region ax.axvspan(growth_start, growth_end, alpha=0.2, color='green', label='Growth Period') # Add threshold line ax.axhline(y=target, color='gray', linestyle='--', label=f'Target: ${target:,.0f}') ax.set_title('Revenue Growth Story', fontsize=14, fontweight='bold') ax.legend() ``` ## Presentation Templates ### Template 1: Executive Summary Slide ``` β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” β”‚ KEY INSIGHT β”‚ β”‚ ══════════════════════════════════════════════════════════│ β”‚ β”‚ β”‚ "Customers who complete onboarding in week 1 β”‚ β”‚ have 3x higher lifetime value" β”‚ β”‚ β”‚ β”œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”¬β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€ β”‚ β”‚ β”‚ β”‚ THE DATA β”‚ THE IMPLICATION β”‚ β”‚ β”‚ β”‚ β”‚ Week 1 completers: β”‚ βœ“ Prioritize onboarding UX β”‚ β”‚ β€’ LTV: $4,500 β”‚ βœ“ Add day-1 success milestones β”‚ β”‚ β€’ Retention: 85% β”‚ βœ“ Proactive week-1 outreach β”‚ β”‚ β€’ NPS: 72 β”‚ β”‚ β”‚ β”‚ Investment: $75K β”‚ β”‚ Others: β”‚ Expected ROI: 8x β”‚ β”‚ β€’ LTV: $1,500 β”‚ β”‚ β”‚ β€’ Retention: 45% β”‚ β”‚ β”‚ β€’ NPS: 34 β”‚ β”‚ β”‚ β”‚ β”‚ β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”΄β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜ ``` ### Template 2: Data Story Flow ``` Slide 1: THE HEADLINE "We can grow 40% faster by fixing onboarding" Slide 2: THE CONTEXT Current state metrics Industry benchmarks Gap analysis Slide 3: THE DISCOVERY What the data revealed Surprising finding Pattern identification Slide 4: THE DEEP DIVE Root cause analysis Segment breakdowns Statistical significance Slide 5: THE RECOMMENDATION Proposed actions Resource requirements Timeline Slide 6: THE IMPACT Expected outcomes ROI calculation Risk assessment Slide 7: THE ASK Specific request Decision needed Next steps ``` ### Template 3: One-Page Dashboard Story ```markdown # Monthly Business Review: January 2024 ## THE HEADLINE Revenue up 15% but CAC increasing faster than LTV ## KEY METRICS AT A GLANCE β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”¬β”€β”€β”€β”€β”€β”€β”€β”€β”¬β”€β”€β”€β”€β”€β”€β”€β”€β”¬β”€β”€β”€β”€β”€β”€β”€β”€β” β”‚ MRR β”‚ NRR β”‚ CAC β”‚ LTV β”‚ β”‚ $125K β”‚ 108% β”‚ $450 β”‚ $2,200 β”‚ β”‚ β–²15% β”‚ β–²3% β”‚ β–²22% β”‚ β–²8% β”‚ β””β”€β”€β”€β”€β”€β”€β”€β”€β”΄β”€β”€β”€β”€β”€β”€β”€β”€β”΄β”€β”€β”€β”€β”€β”€β”€β”€β”΄β”€β”€β”€β”€β”€β”€β”€β”€β”˜ ## WHAT'S WORKING βœ“ Enterprise segment growing 25% MoM βœ“ Referral program driving 30% of new logos βœ“ Support satisfaction at all-time high (94%) ## WHAT NEEDS ATTENTION βœ— SMB acquisition cost up 40% βœ— Trial conversion down 5 points βœ— Time-to-value increased by 3 days ## ROOT CAUSE [Mini chart showing SMB vs Enterprise CAC trend] SMB paid ads becoming less efficient. CPC up 35% while conversion flat. ## RECOMMENDATION 1. Shift $20K/mo from paid to content 2. Launch SMB self-serve trial 3. A/B test shorter onboarding ## NEXT MONTH'S FOCUS - Launch content marketing pilot - Complete self-serve MVP - Reduce time-to-value to < 7 days ``` ## Writing Techniques ### Headlines That Work ```markdown BAD: "Q4 Sales Analysis" GOOD: "Q4 Sales Beat Target by 23% - Here's Why" BAD: "Customer Churn Report" GOOD: "We're Losing $2.4M to Preventable Churn" BAD: "Marketing Performance" GOOD: "Content Marketing Delivers 4x ROI vs. Paid" Formula: [Specific Number] + [Business Impact] + [Actionable Context] ``` ### Transition Phrases ```markdown Building the narrative: β€’ "This leads us to ask..." β€’ "When we dig deeper..." β€’ "The pattern becomes clear when..." β€’ "Contrast this with..." Introducing insights: β€’ "The data reveals..." β€’ "What surprised us was..." β€’ "The inflection point came when..." β€’ "The key finding is..." Moving to action: β€’ "This insight suggests..." β€’ "Based on this analysis..." β€’ "The implication is clear..." β€’ "Our recommendation is..." ``` ### Handling Uncertainty ```markdown Acknowledge limitations: β€’ "With 95% confidence, we can say..." β€’ "The sample size of 500 shows..." β€’ "While correlation is strong, causation requires..." β€’ "This trend holds for [segment], though [caveat]..." Present ranges: β€’ "Impact estimate: $400K-$600K" β€’ "Confidence interval: 15-20% improvement" β€’ "Best case: X, Conservative: Y" ``` ## Best Practices ### Do's - **Start with the "so what"** - Lead with insight - **Use the rule of three** - Three points, three comparisons - **Show, don't tell** - Let data speak - **Make it personal** - Connect to audience goals - **End with action** - Clear next steps ### Don'ts - **Don't data dump** - Curate ruthlessly - **Don't bury the insight** - Front-load key findings - **Don't use jargon** - Match audience vocabulary - **Don't show methodology first** - Context, then method - **Don't forget the narrative** - Numbers need meaning ## Resources - [Storytelling with Data (Cole Nussbaumer)](https://www.storytellingwithdata.com/) - [The Pyramid Principle (Barbara Minto)](https://www.amazon.com/Pyramid-Principle-Logic-Writing-Thinking/dp/0273710516) - [Resonate (Nancy Duarte)](https://www.duarte.com/resonate/)
πŸ‘0
πŸ‘οΈ0
πŸ€– Auto-discovered
πŸ€–system promptβ€’7 months ago

deployment-pipeline-design

Design multi-stage CI/CD pipelines with approval gates, security

coding
⭐1
# Deployment Pipeline Design Architecture patterns for multi-stage CI/CD pipelines with approval gates and deployment strategies. ## Purpose Design robust, secure deployment pipelines that balance speed with safety through proper stage organization and approval workflows. ## When to Use - Design CI/CD architecture - Implement deployment gates - Configure multi-environment pipelines - Establish deployment best practices - Implement progressive delivery ## Pipeline Stages ### Standard Pipeline Flow ``` β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β” β”Œβ”€β”€β”€β”€β”€β”€β” β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β” β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β” β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” β”‚ Build β”‚ β†’ β”‚ Test β”‚ β†’ β”‚ Staging β”‚ β†’ β”‚ Approveβ”‚ β†’ β”‚Productionβ”‚ β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜ β””β”€β”€β”€β”€β”€β”€β”˜ β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜ β””β”€β”€β”€β”€β”€β”€β”€β”€β”˜ β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜ ``` ### Detailed Stage Breakdown 1. **Source** - Code checkout 2. **Build** - Compile, package, containerize 3. **Test** - Unit, integration, security scans 4. **Staging Deploy** - Deploy to staging environment 5. **Integration Tests** - E2E, smoke tests 6. **Approval Gate** - Manual approval required 7. **Production Deploy** - Canary, blue-green, rolling 8. **Verification** - Health checks, monitoring 9. **Rollback** - Automated rollback on failure ## Approval Gate Patterns ### Pattern 1: Manual Approval ```yaml # GitHub Actions production-deploy: needs: staging-deploy environment: name: production url: https://app.example.com runs-on: ubuntu-latest steps: - name: Deploy to production run: | # Deployment commands ``` ### Pattern 2: Time-Based Approval ```yaml # GitLab CI deploy:production: stage: deploy script: - deploy.sh production environment: name: production when: delayed start_in: 30 minutes only: - main ``` ### Pattern 3: Multi-Approver ```yaml # Azure Pipelines stages: - stage: Production dependsOn: Staging jobs: - deployment: Deploy environment: name: production resourceType: Kubernetes strategy: runOnce: preDeploy: steps: - task: ManualValidation@0 inputs: notifyUsers: "team-leads@example.com" instructions: "Review staging metrics before approving" ``` **Reference:** See `assets/approval-gate-template.yml` ## Deployment Strategies ### 1. Rolling Deployment ```yaml apiVersion: apps/v1 kind: Deployment metadata: name: my-app spec: replicas: 10 strategy: type: RollingUpdate rollingUpdate: maxSurge: 2 maxUnavailable: 1 ``` **Characteristics:** - Gradual rollout - Zero downtime - Easy rollback - Best for most applications ### 2. Blue-Green Deployment ```yaml # Blue (current) kubectl apply -f blue-deployment.yaml kubectl label service my-app version=blue # Green (new) kubectl apply -f green-deployment.yaml # Test green environment kubectl label service my-app version=green # Rollback if needed kubectl label service my-app version=blue ``` **Characteristics:** - Instant switchover - Easy rollback - Doubles infrastructure cost temporarily - Good for high-risk deployments ### 3. Canary Deployment ```yaml apiVersion: argoproj.io/v1alpha1 kind: Rollout metadata: name: my-app spec: replicas: 10 strategy: canary: steps: - setWeight: 10 - pause: { duration: 5m } - setWeight: 25 - pause: { duration: 5m } - setWeight: 50 - pause: { duration: 5m } - setWeight: 100 ``` **Characteristics:** - Gradual traffic shift - Risk mitigation - Real user testing - Requires service mesh or similar ### 4. Feature Flags ```python from flagsmith import Flagsmith flagsmith = Flagsmith(environment_key="API_KEY") if flagsmith.has_feature("new_checkout_flow"): # New code path process_checkout_v2() else: # Existing code path process_checkout_v1() ``` **Characteristics:** - Deploy without releasing - A/B testing - Instant rollback - Granular control ## Pipeline Orchestration ### Multi-Stage Pipeline Example ```yaml name: Production Pipeline on: push: branches: [main] jobs: build: runs-on: ubuntu-latest steps: - uses: actions/checkout@v4 - name: Build application run: make build - name: Build Docker image run: docker build -t myapp:${{ github.sha }} . - name: Push to registry run: docker push myapp:${{ github.sha }} test: needs: build runs-on: ubuntu-latest steps: - name: Unit tests run: make test - name: Security scan run: trivy image myapp:${{ github.sha }} deploy-staging: needs: test runs-on: ubuntu-latest environment: name: staging steps: - name: Deploy to staging run: kubectl apply -f k8s/staging/ integration-test: needs: deploy-staging runs-on: ubuntu-latest steps: - name: Run E2E tests run: npm run test:e2e deploy-production: needs: integration-test runs-on: ubuntu-latest environment: name: production steps: - name: Canary deployment run: | kubectl apply -f k8s/production/ kubectl argo rollouts promote my-app verify: needs: deploy-production runs-on: ubuntu-latest steps: - name: Health check run: curl -f https://app.example.com/health - name: Notify team run: | curl -X POST ${{ secrets.SLACK_WEBHOOK }} \ -d '{"text":"Production deployment successful!"}' ``` ## Pipeline Best Practices 1. **Fail fast** - Run quick tests first 2. **Parallel execution** - Run independent jobs concurrently 3. **Caching** - Cache dependencies between runs 4. **Artifact management** - Store build artifacts 5. **Environment parity** - Keep environments consistent 6. **Secrets management** - Use secret stores (Vault, etc.) 7. **Deployment windows** - Schedule deployments appropriately 8. **Monitoring integration** - Track deployment metrics 9. **Rollback automation** - Auto-rollback on failures 10. **Documentation** - Document pipeline stages ## Rollback Strategies ### Automated Rollback ```yaml deploy-and-verify: steps: - name: Deploy new version run: kubectl apply -f k8s/ - name: Wait for rollout run: kubectl rollout status deployment/my-app - name: Health check id: health run: | for i in {1..10}; do if curl -sf https://app.example.com/health; then exit 0 fi sleep 10 done exit 1 - name: Rollback on failure if: failure() run: kubectl rollout undo deployment/my-app ``` ### Manual Rollback ```bash # List revision history kubectl rollout history deployment/my-app # Rollback to previous version kubectl rollout undo deployment/my-app # Rollback to specific revision kubectl rollout undo deployment/my-app --to-revision=3 ``` ## Monitoring and Metrics ### Key Pipeline Metrics - **Deployment Frequency** - How often deployments occur - **Lead Time** - Time from commit to production - **Change Failure Rate** - Percentage of failed deployments - **Mean Time to Recovery (MTTR)** - Time to recover from failure - **Pipeline Success Rate** - Percentage of successful runs - **Average Pipeline Duration** - Time to complete pipeline ### Integration with Monitoring ```yaml - name: Post-deployment verification run: | # Wait for metrics stabilization sleep 60 # Check error rate ERROR_RATE=$(curl -s "$PROMETHEUS_URL/api/v1/query?query=rate(http_errors_total[5m])" | jq '.data.result[0].value[1]') if (( $(echo "$ERROR_RATE > 0.01" | bc -l) )); then echo "Error rate too high: $ERROR_RATE" exit 1 fi ``` ## Reference Files - `references/pipeline-orchestration.md` - Complex pipeline patterns - `assets/approval-gate-template.yml` - Approval workflow templates ## Related Skills - `github-actions-templates` - For GitHub Actions implementation - `gitlab-ci-patterns` - For GitLab CI implementation - `secrets-management` - For secrets handling
πŸ‘0
πŸ‘οΈ0
πŸ€– Auto-discovered
πŸ€–system promptβ€’7 months ago

track-management

Use this skill when creating, managing, or working with Conductor

coding
⭐1
# Track Management Guide for creating, managing, and completing Conductor tracks - the logical work units that organize features, bugs, and refactors through specification, planning, and implementation phases. ## When to Use This Skill - Creating new feature, bug, or refactor tracks - Writing or reviewing spec.md files - Creating or updating plan.md files - Managing track lifecycle from creation to completion - Understanding track status markers and conventions - Working with the tracks.md registry - Interpreting or updating track metadata ## Track Concept A track is a logical work unit that encapsulates a complete piece of work. Each track has: - A unique identifier - A specification defining requirements - A phased plan breaking work into tasks - Metadata tracking status and progress Tracks provide semantic organization for work, enabling: - Clear scope boundaries - Progress tracking - Git-aware operations (revert by track) - Team coordination ## Track Types ### feature New functionality or capabilities. Use for: - New user-facing features - New API endpoints - New integrations - Significant enhancements ### bug Defect fixes. Use for: - Incorrect behavior - Error conditions - Performance regressions - Security vulnerabilities ### chore Maintenance and housekeeping. Use for: - Dependency updates - Configuration changes - Documentation updates - Cleanup tasks ### refactor Code improvement without behavior change. Use for: - Code restructuring - Pattern adoption - Technical debt reduction - Performance optimization (same behavior, better performance) ## Track ID Format Track IDs follow the pattern: `{shortname}_{YYYYMMDD}` - **shortname**: 2-4 word kebab-case description (e.g., `user-auth`, `api-rate-limit`) - **YYYYMMDD**: Creation date in ISO format Examples: - `user-auth_20250115` - `fix-login-error_20250115` - `upgrade-deps_20250115` - `refactor-api-client_20250115` ## Track Lifecycle ### 1. Creation (newTrack) **Define Requirements** 1. Gather requirements through interactive Q&A 2. Identify acceptance criteria 3. Determine scope boundaries 4. Identify dependencies **Generate Specification** 1. Create `spec.md` with structured requirements 2. Document functional and non-functional requirements 3. Define acceptance criteria 4. List dependencies and constraints **Generate Plan** 1. Create `plan.md` with phased task breakdown 2. Organize tasks into logical phases 3. Add verification tasks after phases 4. Estimate effort and complexity **Register Track** 1. Add entry to `tracks.md` registry 2. Create track directory structure 3. Generate `metadata.json` 4. Create track `index.md` ### 2. Implementation **Execute Tasks** 1. Select next pending task from plan 2. Mark task as in-progress 3. Implement following workflow (TDD) 4. Mark task complete with commit SHA **Update Status** 1. Update task markers in plan.md 2. Record commit SHAs for traceability 3. Update phase progress 4. Update track status in tracks.md **Verify Progress** 1. Complete verification tasks 2. Wait for checkpoint approval 3. Record checkpoint commits ### 3. Completion **Sync Documentation** 1. Update product.md if features added 2. Update tech-stack.md if dependencies changed 3. Verify all acceptance criteria met **Archive or Delete** 1. Mark track as completed in tracks.md 2. Record completion date 3. Archive or retain track directory ## Specification (spec.md) Structure ```markdown # {Track Title} ## Overview Brief description of what this track accomplishes and why. ## Functional Requirements ### FR-1: {Requirement Name} Description of the functional requirement. - Acceptance: How to verify this requirement is met ### FR-2: {Requirement Name} ... ## Non-Functional Requirements ### NFR-1: {Requirement Name} Description of the non-functional requirement (performance, security, etc.) - Target: Specific measurable target - Verification: How to test ## Acceptance Criteria - [ ] Criterion 1: Specific, testable condition - [ ] Criterion 2: Specific, testable condition - [ ] Criterion 3: Specific, testable condition ## Scope ### In Scope - Explicitly included items - Features to implement - Components to modify ### Out of Scope - Explicitly excluded items - Future considerations - Related but separate work ## Dependencies ### Internal - Other tracks or components this depends on - Required context artifacts ### External - Third-party services or APIs - External dependencies ## Risks and Mitigations | Risk | Impact | Mitigation | | ---------------- | --------------- | ------------------- | | Risk description | High/Medium/Low | Mitigation strategy | ## Open Questions - [ ] Question that needs resolution - [x] Resolved question - Answer ``` ## Plan (plan.md) Structure ```markdown # Implementation Plan: {Track Title} Track ID: `{track-id}` Created: YYYY-MM-DD Status: pending | in-progress | completed ## Overview Brief description of implementation approach. ## Phase 1: {Phase Name} ### Tasks - [ ] **Task 1.1**: Task description - Sub-task or detail - Sub-task or detail - [ ] **Task 1.2**: Task description - [ ] **Task 1.3**: Task description ### Verification - [ ] **Verify 1.1**: Verification step for phase ## Phase 2: {Phase Name} ### Tasks - [ ] **Task 2.1**: Task description - [ ] **Task 2.2**: Task description ### Verification - [ ] **Verify 2.1**: Verification step for phase ## Phase 3: Finalization ### Tasks - [ ] **Task 3.1**: Update documentation - [ ] **Task 3.2**: Final integration test ### Verification - [ ] **Verify 3.1**: All acceptance criteria met ## Checkpoints | Phase | Checkpoint SHA | Date | Status | | ------- | -------------- | ---- | ------- | | Phase 1 | | | pending | | Phase 2 | | | pending | | Phase 3 | | | pending | ``` ## Status Marker Conventions Use consistent markers in plan.md: | Marker | Meaning | Usage | | ------ | ----------- | --------------------------- | | `[ ]` | Pending | Task not started | | `[~]` | In Progress | Currently being worked | | `[x]` | Complete | Task finished (include SHA) | | `[-]` | Skipped | Intentionally not done | | `[!]` | Blocked | Waiting on dependency | Example: ```markdown - [x] **Task 1.1**: Set up database schema `abc1234` - [~] **Task 1.2**: Implement user model - [ ] **Task 1.3**: Add validation logic - [!] **Task 1.4**: Integrate auth service (blocked: waiting for API key) - [-] **Task 1.5**: Legacy migration (skipped: not needed) ``` ## Track Registry (tracks.md) Format ```markdown # Track Registry ## Active Tracks | Track ID | Type | Status | Phase | Started | Assignee | | ------------------------------------------------ | ------- | ----------- | ----- | ---------- | ---------- | | [user-auth_20250115](tracks/user-auth_20250115/) | feature | in-progress | 2/3 | 2025-01-15 | @developer | | [fix-login_20250114](tracks/fix-login_20250114/) | bug | pending | 0/2 | 2025-01-14 | - | ## Completed Tracks | Track ID | Type | Completed | Duration | | ---------------------------------------------- | ----- | ---------- | -------- | | [setup-ci_20250110](tracks/setup-ci_20250110/) | chore | 2025-01-12 | 2 days | ## Archived Tracks | Track ID | Reason | Archived | | ---------------------------------------------------- | ---------- | ---------- | | [old-feature_20241201](tracks/old-feature_20241201/) | Superseded | 2025-01-05 | ``` ## Metadata (metadata.json) Fields ```json { "id": "user-auth_20250115", "title": "User Authentication System", "type": "feature", "status": "in-progress", "priority": "high", "created": "2025-01-15T10:30:00Z", "updated": "2025-01-15T14:45:00Z", "started": "2025-01-15T11:00:00Z", "completed": null, "assignee": "@developer", "phases": { "total": 3, "current": 2, "completed": 1 }, "tasks": { "total": 12, "completed": 5, "in_progress": 1, "pending": 6 }, "checkpoints": [ { "phase": 1, "sha": "abc1234", "date": "2025-01-15T13:00:00Z" } ], "dependencies": [], "tags": ["auth", "security"] } ``` ## Track Operations ### Creating a Track 1. Run `/conductor:new-track` 2. Answer interactive questions 3. Review generated spec.md 4. Review generated plan.md 5. Confirm track creation ### Starting Implementation 1. Read spec.md and plan.md 2. Verify context artifacts are current 3. Mark first task as `[~]` 4. Begin TDD workflow ### Completing a Phase 1. Ensure all phase tasks are `[x]` 2. Complete verification tasks 3. Wait for checkpoint approval 4. Record checkpoint SHA 5. Proceed to next phase ### Completing a Track 1. Verify all phases complete 2. Verify all acceptance criteria met 3. Update product.md if needed 4. Mark track completed in tracks.md 5. Update metadata.json ### Reverting a Track 1. Run `/conductor:revert` 2. Select track to revert 3. Choose granularity (track/phase/task) 4. Confirm revert operation 5. Update status markers ## Handling Track Dependencies ### Identifying Dependencies During track creation, identify: - **Hard dependencies**: Must complete before this track can start - **Soft dependencies**: Can proceed in parallel but may affect integration - **External dependencies**: Third-party services, APIs, or team decisions ### Documenting Dependencies In spec.md, list dependencies with: - Dependency type (hard/soft/external) - Current status (available/pending/blocked) - Resolution path (what needs to happen) ### Managing Blocked Tracks When a track is blocked: 1. Mark blocked tasks with `[!]` and reason 2. Update tracks.md status 3. Document blocker in metadata.json 4. Consider creating dependency track if needed ## Track Sizing Guidelines ### Right-Sized Tracks Aim for tracks that: - Complete in 1-5 days of work - Have 2-4 phases - Contain 8-20 tasks total - Deliver a coherent, testable unit ### Too Large Signs a track is too large: - More than 5 phases - More than 25 tasks - Multiple unrelated features - Estimated duration > 1 week Solution: Split into multiple tracks with clear boundaries. ### Too Small Signs a track is too small: - Single phase with 1-2 tasks - No meaningful verification needed - Could be a sub-task of another track - Less than a few hours of work Solution: Combine with related work or handle as part of existing track. ## Specification Quality Checklist Before finalizing spec.md, verify: ### Requirements Quality - [ ] Each requirement has clear acceptance criteria - [ ] Requirements are testable - [ ] Requirements are independent (can verify separately) - [ ] No ambiguous language ("should be fast" β†’ "response < 200ms") ### Scope Clarity - [ ] In-scope items are specific - [ ] Out-of-scope items prevent scope creep - [ ] Boundaries are clear to implementer ### Dependencies Identified - [ ] All internal dependencies listed - [ ] External dependencies have owners/contacts - [ ] Dependency status is current ### Risks Addressed - [ ] Major risks identified - [ ] Impact assessment realistic - [ ] Mitigations are actionable ## Plan Quality Checklist Before starting implementation, verify plan.md: ### Task Quality - [ ] Tasks are atomic (one logical action) - [ ] Tasks are independently verifiable - [ ] Task descriptions are clear - [ ] Sub-tasks provide helpful detail ### Phase Organization - [ ] Phases group related tasks - [ ] Each phase delivers something testable - [ ] Verification tasks after each phase - [ ] Phases build on each other logically ### Completeness - [ ] All spec requirements have corresponding tasks - [ ] Documentation tasks included - [ ] Testing tasks included - [ ] Integration tasks included ## Common Track Patterns ### Feature Track Pattern ``` Phase 1: Foundation - Data models - Database migrations - Basic API structure Phase 2: Core Logic - Business logic implementation - Input validation - Error handling Phase 3: Integration - UI integration - API documentation - End-to-end tests ``` ### Bug Fix Track Pattern ``` Phase 1: Reproduction - Write failing test capturing bug - Document reproduction steps Phase 2: Fix - Implement fix - Verify test passes - Check for regressions Phase 3: Verification - Manual verification - Update documentation if needed ``` ### Refactor Track Pattern ``` Phase 1: Preparation - Add characterization tests - Document current behavior Phase 2: Refactoring - Apply changes incrementally - Maintain green tests throughout Phase 3: Cleanup - Remove dead code - Update documentation ``` ## Best Practices 1. **One track, one concern**: Keep tracks focused on a single logical change 2. **Small phases**: Break work into phases of 3-5 tasks maximum 3. **Verification after phases**: Always include verification tasks 4. **Update markers immediately**: Mark task status as you work 5. **Record SHAs**: Always note commit SHAs for completed tasks 6. **Review specs before planning**: Ensure spec is complete before creating plan 7. **Link dependencies**: Explicitly note track dependencies 8. **Archive, don't delete**: Preserve completed tracks for reference 9. **Size appropriately**: Keep tracks between 1-5 days of work 10. **Clear acceptance criteria**: Every requirement must be testable
πŸ‘0
πŸ‘οΈ0
πŸ€– Auto-discovered
πŸ€–system promptβ€’7 months ago

architecture-decision-records

Write and maintain Architecture Decision Records (ADRs) following

coding
⭐1
# Architecture Decision Records Comprehensive patterns for creating, maintaining, and managing Architecture Decision Records (ADRs) that capture the context and rationale behind significant technical decisions. ## When to Use This Skill - Making significant architectural decisions - Documenting technology choices - Recording design trade-offs - Onboarding new team members - Reviewing historical decisions - Establishing decision-making processes ## Core Concepts ### 1. What is an ADR? An Architecture Decision Record captures: - **Context**: Why we needed to make a decision - **Decision**: What we decided - **Consequences**: What happens as a result ### 2. When to Write an ADR | Write ADR | Skip ADR | | -------------------------- | ---------------------- | | New framework adoption | Minor version upgrades | | Database technology choice | Bug fixes | | API design patterns | Implementation details | | Security architecture | Routine maintenance | | Integration patterns | Configuration changes | ### 3. ADR Lifecycle ``` Proposed β†’ Accepted β†’ Deprecated β†’ Superseded ↓ Rejected ``` ## Templates ### Template 1: Standard ADR (MADR Format) ```markdown # ADR-0001: Use PostgreSQL as Primary Database ## Status Accepted ## Context We need to select a primary database for our new e-commerce platform. The system will handle: - ~10,000 concurrent users - Complex product catalog with hierarchical categories - Transaction processing for orders and payments - Full-text search for products - Geospatial queries for store locator The team has experience with MySQL, PostgreSQL, and MongoDB. We need ACID compliance for financial transactions. ## Decision Drivers - **Must have ACID compliance** for payment processing - **Must support complex queries** for reporting - **Should support full-text search** to reduce infrastructure complexity - **Should have good JSON support** for flexible product attributes - **Team familiarity** reduces onboarding time ## Considered Options ### Option 1: PostgreSQL - **Pros**: ACID compliant, excellent JSON support (JSONB), built-in full-text search, PostGIS for geospatial, team has experience - **Cons**: Slightly more complex replication setup than MySQL ### Option 2: MySQL - **Pros**: Very familiar to team, simple replication, large community - **Cons**: Weaker JSON support, no built-in full-text search (need Elasticsearch), no geospatial without extensions ### Option 3: MongoDB - **Pros**: Flexible schema, native JSON, horizontal scaling - **Cons**: No ACID for multi-document transactions (at decision time), team has limited experience, requires schema design discipline ## Decision We will use **PostgreSQL 15** as our primary database. ## Rationale PostgreSQL provides the best balance of: 1. **ACID compliance** essential for e-commerce transactions 2. **Built-in capabilities** (full-text search, JSONB, PostGIS) reduce infrastructure complexity 3. **Team familiarity** with SQL databases reduces learning curve 4. **Mature ecosystem** with excellent tooling and community support The slight complexity in replication is outweighed by the reduction in additional services (no separate Elasticsearch needed). ## Consequences ### Positive - Single database handles transactions, search, and geospatial queries - Reduced operational complexity (fewer services to manage) - Strong consistency guarantees for financial data - Team can leverage existing SQL expertise ### Negative - Need to learn PostgreSQL-specific features (JSONB, full-text search syntax) - Vertical scaling limits may require read replicas sooner - Some team members need PostgreSQL-specific training ### Risks - Full-text search may not scale as well as dedicated search engines - Mitigation: Design for potential Elasticsearch addition if needed ## Implementation Notes - Use JSONB for flexible product attributes - Implement connection pooling with PgBouncer - Set up streaming replication for read replicas - Use pg_trgm extension for fuzzy search ## Related Decisions - ADR-0002: Caching Strategy (Redis) - complements database choice - ADR-0005: Search Architecture - may supersede if Elasticsearch needed ## References - [PostgreSQL JSON Documentation](https://www.postgresql.org/docs/current/datatype-json.html) - [PostgreSQL Full Text Search](https://www.postgresql.org/docs/current/textsearch.html) - Internal: Performance benchmarks in `/docs/benchmarks/database-comparison.md` ``` ### Template 2: Lightweight ADR ```markdown # ADR-0012: Adopt TypeScript for Frontend Development **Status**: Accepted **Date**: 2024-01-15 **Deciders**: @alice, @bob, @charlie ## Context Our React codebase has grown to 50+ components with increasing bug reports related to prop type mismatches and undefined errors. PropTypes provide runtime-only checking. ## Decision Adopt TypeScript for all new frontend code. Migrate existing code incrementally. ## Consequences **Good**: Catch type errors at compile time, better IDE support, self-documenting code. **Bad**: Learning curve for team, initial slowdown, build complexity increase. **Mitigations**: TypeScript training sessions, allow gradual adoption with `allowJs: true`. ``` ### Template 3: Y-Statement Format ```markdown # ADR-0015: API Gateway Selection In the context of **building a microservices architecture**, facing **the need for centralized API management, authentication, and rate limiting**, we decided for **Kong Gateway** and against **AWS API Gateway and custom Nginx solution**, to achieve **vendor independence, plugin extensibility, and team familiarity with Lua**, accepting that **we need to manage Kong infrastructure ourselves**. ``` ### Template 4: ADR for Deprecation ```markdown # ADR-0020: Deprecate MongoDB in Favor of PostgreSQL ## Status Accepted (Supersedes ADR-0003) ## Context ADR-0003 (2021) chose MongoDB for user profile storage due to schema flexibility needs. Since then: - MongoDB's multi-document transactions remain problematic for our use case - Our schema has stabilized and rarely changes - We now have PostgreSQL expertise from other services - Maintaining two databases increases operational burden ## Decision Deprecate MongoDB and migrate user profiles to PostgreSQL. ## Migration Plan 1. **Phase 1** (Week 1-2): Create PostgreSQL schema, dual-write enabled 2. **Phase 2** (Week 3-4): Backfill historical data, validate consistency 3. **Phase 3** (Week 5): Switch reads to PostgreSQL, monitor 4. **Phase 4** (Week 6): Remove MongoDB writes, decommission ## Consequences ### Positive - Single database technology reduces operational complexity - ACID transactions for user data - Team can focus PostgreSQL expertise ### Negative - Migration effort (~4 weeks) - Risk of data issues during migration - Lose some schema flexibility ## Lessons Learned Document from ADR-0003 experience: - Schema flexibility benefits were overestimated - Operational cost of multiple databases was underestimated - Consider long-term maintenance in technology decisions ``` ### Template 5: Request for Comments (RFC) Style ```markdown # RFC-0025: Adopt Event Sourcing for Order Management ## Summary Propose adopting event sourcing pattern for the order management domain to improve auditability, enable temporal queries, and support business analytics. ## Motivation Current challenges: 1. Audit requirements need complete order history 2. "What was the order state at time X?" queries are impossible 3. Analytics team needs event stream for real-time dashboards 4. Order state reconstruction for customer support is manual ## Detailed Design ### Event Store ``` OrderCreated { orderId, customerId, items[], timestamp } OrderItemAdded { orderId, item, timestamp } OrderItemRemoved { orderId, itemId, timestamp } PaymentReceived { orderId, amount, paymentId, timestamp } OrderShipped { orderId, trackingNumber, timestamp } ``` ### Projections - **CurrentOrderState**: Materialized view for queries - **OrderHistory**: Complete timeline for audit - **DailyOrderMetrics**: Analytics aggregation ### Technology - Event Store: EventStoreDB (purpose-built, handles projections) - Alternative considered: Kafka + custom projection service ## Drawbacks - Learning curve for team - Increased complexity vs. CRUD - Need to design events carefully (immutable once stored) - Storage growth (events never deleted) ## Alternatives 1. **Audit tables**: Simpler but doesn't enable temporal queries 2. **CDC from existing DB**: Complex, doesn't change data model 3. **Hybrid**: Event source only for order state changes ## Unresolved Questions - [ ] Event schema versioning strategy - [ ] Retention policy for events - [ ] Snapshot frequency for performance ## Implementation Plan 1. Prototype with single order type (2 weeks) 2. Team training on event sourcing (1 week) 3. Full implementation and migration (4 weeks) 4. Monitoring and optimization (ongoing) ## References - [Event Sourcing by Martin Fowler](https://martinfowler.com/eaaDev/EventSourcing.html) - [EventStoreDB Documentation](https://www.eventstore.com/docs) ``` ## ADR Management ### Directory Structure ``` docs/ β”œβ”€β”€ adr/ β”‚ β”œβ”€β”€ README.md # Index and guidelines β”‚ β”œβ”€β”€ template.md # Team's ADR template β”‚ β”œβ”€β”€ 0001-use-postgresql.md β”‚ β”œβ”€β”€ 0002-caching-strategy.md β”‚ β”œβ”€β”€ 0003-mongodb-user-profiles.md # [DEPRECATED] β”‚ └── 0020-deprecate-mongodb.md # Supersedes 0003 ``` ### ADR Index (README.md) ```markdown # Architecture Decision Records This directory contains Architecture Decision Records (ADRs) for [Project Name]. ## Index | ADR | Title | Status | Date | | ------------------------------------- | ---------------------------------- | ---------- | ---------- | | [0001](0001-use-postgresql.md) | Use PostgreSQL as Primary Database | Accepted | 2024-01-10 | | [0002](0002-caching-strategy.md) | Caching Strategy with Redis | Accepted | 2024-01-12 | | [0003](0003-mongodb-user-profiles.md) | MongoDB for User Profiles | Deprecated | 2023-06-15 | | [0020](0020-deprecate-mongodb.md) | Deprecate MongoDB | Accepted | 2024-01-15 | ## Creating a New ADR 1. Copy `template.md` to `NNNN-title-with-dashes.md` 2. Fill in the template 3. Submit PR for review 4. Update this index after approval ## ADR Status - **Proposed**: Under discussion - **Accepted**: Decision made, implementing - **Deprecated**: No longer relevant - **Superseded**: Replaced by another ADR - **Rejected**: Considered but not adopted ``` ### Automation (adr-tools) ```bash # Install adr-tools brew install adr-tools # Initialize ADR directory adr init docs/adr # Create new ADR adr new "Use PostgreSQL as Primary Database" # Supersede an ADR adr new -s 3 "Deprecate MongoDB in Favor of PostgreSQL" # Generate table of contents adr generate toc > docs/adr/README.md # Link related ADRs adr link 2 "Complements" 1 "Is complemented by" ``` ## Review Process ```markdown ## ADR Review Checklist ### Before Submission - [ ] Context clearly explains the problem - [ ] All viable options considered - [ ] Pros/cons balanced and honest - [ ] Consequences (positive and negative) documented - [ ] Related ADRs linked ### During Review - [ ] At least 2 senior engineers reviewed - [ ] Affected teams consulted - [ ] Security implications considered - [ ] Cost implications documented - [ ] Reversibility assessed ### After Acceptance - [ ] ADR index updated - [ ] Team notified - [ ] Implementation tickets created - [ ] Related documentation updated ``` ## Best Practices ### Do's - **Write ADRs early** - Before implementation starts - **Keep them short** - 1-2 pages maximum - **Be honest about trade-offs** - Include real cons - **Link related decisions** - Build decision graph - **Update status** - Deprecate when superseded ### Don'ts - **Don't change accepted ADRs** - Write new ones to supersede - **Don't skip context** - Future readers need background - **Don't hide failures** - Rejected decisions are valuable - **Don't be vague** - Specific decisions, specific consequences - **Don't forget implementation** - ADR without action is waste ## Resources - [Documenting Architecture Decisions (Michael Nygard)](https://cognitect.com/blog/2011/11/15/documenting-architecture-decisions) - [MADR Template](https://adr.github.io/madr/) - [ADR GitHub Organization](https://adr.github.io/) - [adr-tools](https://github.com/npryce/adr-tools)
πŸ‘0
πŸ‘οΈ0
πŸ€– Auto-discovered
πŸ€–system promptβ€’7 months ago

incident-runbook-templates

Create structured incident response runbooks with step-by-step

coding
⭐1
# Incident Runbook Templates Production-ready templates for incident response runbooks covering detection, triage, mitigation, resolution, and communication. ## When to Use This Skill - Creating incident response procedures - Building service-specific runbooks - Establishing escalation paths - Documenting recovery procedures - Responding to active incidents - Onboarding on-call engineers ## Core Concepts ### 1. Incident Severity Levels | Severity | Impact | Response Time | Example | | -------- | -------------------------- | ----------------- | ----------------------- | | **SEV1** | Complete outage, data loss | 15 min | Production down | | **SEV2** | Major degradation | 30 min | Critical feature broken | | **SEV3** | Minor impact | 2 hours | Non-critical bug | | **SEV4** | Minimal impact | Next business day | Cosmetic issue | ### 2. Runbook Structure ``` 1. Overview & Impact 2. Detection & Alerts 3. Initial Triage 4. Mitigation Steps 5. Root Cause Investigation 6. Resolution Procedures 7. Verification & Rollback 8. Communication Templates 9. Escalation Matrix ``` ## Runbook Templates ### Template 1: Service Outage Runbook ````markdown # [Service Name] Outage Runbook ## Overview **Service**: Payment Processing Service **Owner**: Platform Team **Slack**: #payments-incidents **PagerDuty**: payments-oncall ## Impact Assessment - [ ] Which customers are affected? - [ ] What percentage of traffic is impacted? - [ ] Are there financial implications? - [ ] What's the blast radius? ## Detection ### Alerts - `payment_error_rate > 5%` (PagerDuty) - `payment_latency_p99 > 2s` (Slack) - `payment_success_rate < 95%` (PagerDuty) ### Dashboards - [Payment Service Dashboard](https://grafana/d/payments) - [Error Tracking](https://sentry.io/payments) - [Dependency Status](https://status.stripe.com) ## Initial Triage (First 5 Minutes) ### 1. Assess Scope ```bash # Check service health kubectl get pods -n payments -l app=payment-service # Check recent deployments kubectl rollout history deployment/payment-service -n payments # Check error rates curl -s "http://prometheus:9090/api/v1/query?query=sum(rate(http_requests_total{status=~'5..'}[5m]))" ``` ```` ### 2. Quick Health Checks - [ ] Can you reach the service? `curl -I https://api.company.com/payments/health` - [ ] Database connectivity? Check connection pool metrics - [ ] External dependencies? Check Stripe, bank API status - [ ] Recent changes? Check deploy history ### 3. Initial Classification | Symptom | Likely Cause | Go To Section | | -------------------- | ------------------- | ------------- | | All requests failing | Service down | Section 4.1 | | High latency | Database/dependency | Section 4.2 | | Partial failures | Code bug | Section 4.3 | | Spike in errors | Traffic surge | Section 4.4 | ## Mitigation Procedures ### 4.1 Service Completely Down ```bash # Step 1: Check pod status kubectl get pods -n payments # Step 2: If pods are crash-looping, check logs kubectl logs -n payments -l app=payment-service --tail=100 # Step 3: Check recent deployments kubectl rollout history deployment/payment-service -n payments # Step 4: ROLLBACK if recent deploy is suspect kubectl rollout undo deployment/payment-service -n payments # Step 5: Scale up if resource constrained kubectl scale deployment/payment-service -n payments --replicas=10 # Step 6: Verify recovery kubectl rollout status deployment/payment-service -n payments ``` ### 4.2 High Latency ```bash # Step 1: Check database connections kubectl exec -n payments deploy/payment-service -- \ curl localhost:8080/metrics | grep db_pool # Step 2: Check slow queries (if DB issue) psql -h $DB_HOST -U $DB_USER -c " SELECT pid, now() - query_start AS duration, query FROM pg_stat_activity WHERE state = 'active' AND duration > interval '5 seconds' ORDER BY duration DESC;" # Step 3: Kill long-running queries if needed psql -h $DB_HOST -U $DB_USER -c "SELECT pg_terminate_backend(pid);" # Step 4: Check external dependency latency curl -w "@curl-format.txt" -o /dev/null -s https://api.stripe.com/v1/health # Step 5: Enable circuit breaker if dependency is slow kubectl set env deployment/payment-service \ STRIPE_CIRCUIT_BREAKER_ENABLED=true -n payments ``` ### 4.3 Partial Failures (Specific Errors) ```bash # Step 1: Identify error pattern kubectl logs -n payments -l app=payment-service --tail=500 | \ grep -i error | sort | uniq -c | sort -rn | head -20 # Step 2: Check error tracking # Go to Sentry: https://sentry.io/payments # Step 3: If specific endpoint, enable feature flag to disable curl -X POST https://api.company.com/internal/feature-flags \ -d '{"flag": "DISABLE_PROBLEMATIC_FEATURE", "enabled": true}' # Step 4: If data issue, check recent data changes psql -h $DB_HOST -c " SELECT * FROM audit_log WHERE table_name = 'payment_methods' AND created_at > now() - interval '1 hour';" ``` ### 4.4 Traffic Surge ```bash # Step 1: Check current request rate kubectl top pods -n payments # Step 2: Scale horizontally kubectl scale deployment/payment-service -n payments --replicas=20 # Step 3: Enable rate limiting kubectl set env deployment/payment-service \ RATE_LIMIT_ENABLED=true \ RATE_LIMIT_RPS=1000 -n payments # Step 4: If attack, block suspicious IPs kubectl apply -f - <<EOF apiVersion: networking.k8s.io/v1 kind: NetworkPolicy metadata: name: block-suspicious namespace: payments spec: podSelector: matchLabels: app: payment-service ingress: - from: - ipBlock: cidr: 0.0.0.0/0 except: - 192.168.1.0/24 # Suspicious range EOF ``` ## Verification Steps ```bash # Verify service is healthy curl -s https://api.company.com/payments/health | jq # Verify error rate is back to normal curl -s "http://prometheus:9090/api/v1/query?query=sum(rate(http_requests_total{status=~'5..'}[5m]))" | jq '.data.result[0].value[1]' # Verify latency is acceptable curl -s "http://prometheus:9090/api/v1/query?query=histogram_quantile(0.99,sum(rate(http_request_duration_seconds_bucket[5m]))by(le))" | jq # Smoke test critical flows ./scripts/smoke-test-payments.sh ``` ## Rollback Procedures ```bash # Rollback Kubernetes deployment kubectl rollout undo deployment/payment-service -n payments # Rollback database migration (if applicable) ./scripts/db-rollback.sh $MIGRATION_VERSION # Rollback feature flag curl -X POST https://api.company.com/internal/feature-flags \ -d '{"flag": "NEW_PAYMENT_FLOW", "enabled": false}' ``` ## Escalation Matrix | Condition | Escalate To | Contact | | ----------------------------- | ------------------- | ------------------- | | > 15 min unresolved SEV1 | Engineering Manager | @manager (Slack) | | Data breach suspected | Security Team | #security-incidents | | Financial impact > $10k | Finance + Legal | @finance-oncall | | Customer communication needed | Support Lead | @support-lead | ## Communication Templates ### Initial Notification (Internal) ``` 🚨 INCIDENT: Payment Service Degradation Severity: SEV2 Status: Investigating Impact: ~20% of payment requests failing Start Time: [TIME] Incident Commander: [NAME] Current Actions: - Investigating root cause - Scaling up service - Monitoring dashboards Updates in #payments-incidents ``` ### Status Update ``` πŸ“Š UPDATE: Payment Service Incident Status: Mitigating Impact: Reduced to ~5% failure rate Duration: 25 minutes Actions Taken: - Rolled back deployment v2.3.4 β†’ v2.3.3 - Scaled service from 5 β†’ 10 replicas Next Steps: - Continuing to monitor - Root cause analysis in progress ETA to Resolution: ~15 minutes ``` ### Resolution Notification ``` βœ… RESOLVED: Payment Service Incident Duration: 45 minutes Impact: ~5,000 affected transactions Root Cause: Memory leak in v2.3.4 Resolution: - Rolled back to v2.3.3 - Transactions auto-retried successfully Follow-up: - Postmortem scheduled for [DATE] - Bug fix in progress ``` ```` ### Template 2: Database Incident Runbook ```markdown # Database Incident Runbook ## Quick Reference | Issue | Command | |-------|---------| | Check connections | `SELECT count(*) FROM pg_stat_activity;` | | Kill query | `SELECT pg_terminate_backend(pid);` | | Check replication lag | `SELECT extract(epoch from (now() - pg_last_xact_replay_timestamp()));` | | Check locks | `SELECT * FROM pg_locks WHERE NOT granted;` | ## Connection Pool Exhaustion ```sql -- Check current connections SELECT datname, usename, state, count(*) FROM pg_stat_activity GROUP BY datname, usename, state ORDER BY count(*) DESC; -- Identify long-running connections SELECT pid, usename, datname, state, query_start, query FROM pg_stat_activity WHERE state != 'idle' ORDER BY query_start; -- Terminate idle connections SELECT pg_terminate_backend(pid) FROM pg_stat_activity WHERE state = 'idle' AND query_start < now() - interval '10 minutes'; ```` ## Replication Lag ```sql -- Check lag on replica SELECT CASE WHEN pg_last_wal_receive_lsn() = pg_last_wal_replay_lsn() THEN 0 ELSE extract(epoch from now() - pg_last_xact_replay_timestamp()) END AS lag_seconds; -- If lag > 60s, consider: -- 1. Check network between primary/replica -- 2. Check replica disk I/O -- 3. Consider failover if unrecoverable ``` ## Disk Space Critical ```bash # Check disk usage df -h /var/lib/postgresql/data # Find large tables psql -c "SELECT relname, pg_size_pretty(pg_total_relation_size(relid)) FROM pg_catalog.pg_statio_user_tables ORDER BY pg_total_relation_size(relid) DESC LIMIT 10;" # VACUUM to reclaim space psql -c "VACUUM FULL large_table;" # If emergency, delete old data or expand disk ``` ``` ## Best Practices ### Do's - **Keep runbooks updated** - Review after every incident - **Test runbooks regularly** - Game days, chaos engineering - **Include rollback steps** - Always have an escape hatch - **Document assumptions** - What must be true for steps to work - **Link to dashboards** - Quick access during stress ### Don'ts - **Don't assume knowledge** - Write for 3 AM brain - **Don't skip verification** - Confirm each step worked - **Don't forget communication** - Keep stakeholders informed - **Don't work alone** - Escalate early - **Don't skip postmortems** - Learn from every incident ## Resources - [Google SRE Book - Incident Management](https://sre.google/sre-book/managing-incidents/) - [PagerDuty Incident Response](https://response.pagerduty.com/) - [Atlassian Incident Management](https://www.atlassian.com/incident-management) ```
πŸ‘0
πŸ‘οΈ0
πŸ€– Auto-discovered
πŸ€–system promptβ€’7 months ago

on-call-handoff-patterns

Master on-call shift handoffs with context transfer, escalation

coding
⭐1
# On-Call Handoff Patterns Effective patterns for on-call shift transitions, ensuring continuity, context transfer, and reliable incident response across shifts. ## When to Use This Skill - Transitioning on-call responsibilities - Writing shift handoff summaries - Documenting ongoing investigations - Establishing on-call rotation procedures - Improving handoff quality - Onboarding new on-call engineers ## Core Concepts ### 1. Handoff Components | Component | Purpose | | -------------------------- | ----------------------- | | **Active Incidents** | What's currently broken | | **Ongoing Investigations** | Issues being debugged | | **Recent Changes** | Deployments, configs | | **Known Issues** | Workarounds in place | | **Upcoming Events** | Maintenance, releases | ### 2. Handoff Timing ``` Recommended: 30 min overlap between shifts Outgoing: β”œβ”€β”€ 15 min: Write handoff document └── 15 min: Sync call with incoming Incoming: β”œβ”€β”€ 15 min: Review handoff document β”œβ”€β”€ 15 min: Sync call with outgoing └── 5 min: Verify alerting setup ``` ## Templates ### Template 1: Shift Handoff Document ````markdown # On-Call Handoff: Platform Team **Outgoing**: @alice (2024-01-15 to 2024-01-22) **Incoming**: @bob (2024-01-22 to 2024-01-29) **Handoff Time**: 2024-01-22 09:00 UTC --- ## πŸ”΄ Active Incidents ### None currently active No active incidents at handoff time. --- ## 🟑 Ongoing Investigations ### 1. Intermittent API Timeouts (ENG-1234) **Status**: Investigating **Started**: 2024-01-20 **Impact**: ~0.1% of requests timing out **Context**: - Timeouts correlate with database backup window (02:00-03:00 UTC) - Suspect backup process causing lock contention - Added extra logging in PR #567 (deployed 01/21) **Next Steps**: - [ ] Review new logs after tonight's backup - [ ] Consider moving backup window if confirmed **Resources**: - Dashboard: [API Latency](https://grafana/d/api-latency) - Thread: #platform-eng (01/20, 14:32) --- ### 2. Memory Growth in Auth Service (ENG-1235) **Status**: Monitoring **Started**: 2024-01-18 **Impact**: None yet (proactive) **Context**: - Memory usage growing ~5% per day - No memory leak found in profiling - Suspect connection pool not releasing properly **Next Steps**: - [ ] Review heap dump from 01/21 - [ ] Consider restart if usage > 80% **Resources**: - Dashboard: [Auth Service Memory](https://grafana/d/auth-memory) - Analysis doc: [Memory Investigation](https://docs/eng-1235) --- ## 🟒 Resolved This Shift ### Payment Service Outage (2024-01-19) - **Duration**: 23 minutes - **Root Cause**: Database connection exhaustion - **Resolution**: Rolled back v2.3.4, increased pool size - **Postmortem**: [POSTMORTEM-89](https://docs/postmortem-89) - **Follow-up tickets**: ENG-1230, ENG-1231 --- ## πŸ“‹ Recent Changes ### Deployments | Service | Version | Time | Notes | | ------------ | ------- | ----------- | -------------------------- | | api-gateway | v3.2.1 | 01/21 14:00 | Bug fix for header parsing | | user-service | v2.8.0 | 01/20 10:00 | New profile features | | auth-service | v4.1.2 | 01/19 16:00 | Security patch | ### Configuration Changes - 01/21: Increased API rate limit from 1000 to 1500 RPS - 01/20: Updated database connection pool max from 50 to 75 ### Infrastructure - 01/20: Added 2 nodes to Kubernetes cluster - 01/19: Upgraded Redis from 6.2 to 7.0 --- ## ⚠️ Known Issues & Workarounds ### 1. Slow Dashboard Loading **Issue**: Grafana dashboards slow on Monday mornings **Workaround**: Wait 5 min after 08:00 UTC for cache warm-up **Ticket**: OPS-456 (P3) ### 2. Flaky Integration Test **Issue**: `test_payment_flow` fails intermittently in CI **Workaround**: Re-run failed job (usually passes on retry) **Ticket**: ENG-1200 (P2) --- ## πŸ“… Upcoming Events | Date | Event | Impact | Contact | | ----------- | -------------------- | ------------------- | ------------- | | 01/23 02:00 | Database maintenance | 5 min read-only | @dba-team | | 01/24 14:00 | Major release v5.0 | Monitor closely | @release-team | | 01/25 | Marketing campaign | 2x traffic expected | @platform | --- ## πŸ“ž Escalation Reminders | Issue Type | First Escalation | Second Escalation | | --------------- | -------------------- | ----------------- | | Payment issues | @payments-oncall | @payments-manager | | Auth issues | @auth-oncall | @security-team | | Database issues | @dba-team | @infra-manager | | Unknown/severe | @engineering-manager | @vp-engineering | --- ## πŸ”§ Quick Reference ### Common Commands ```bash # Check service health kubectl get pods -A | grep -v Running # Recent deployments kubectl get events --sort-by='.lastTimestamp' | tail -20 # Database connections psql -c "SELECT count(*) FROM pg_stat_activity;" # Clear cache (emergency only) redis-cli FLUSHDB ``` ```` ### Important Links - [Runbooks](https://wiki/runbooks) - [Service Catalog](https://wiki/services) - [Incident Slack](https://slack.com/incidents) - [PagerDuty](https://pagerduty.com/schedules) --- ## Handoff Checklist ### Outgoing Engineer - [x] Document active incidents - [x] Document ongoing investigations - [x] List recent changes - [x] Note known issues - [x] Add upcoming events - [x] Sync with incoming engineer ### Incoming Engineer - [ ] Read this document - [ ] Join sync call - [ ] Verify PagerDuty is routing to you - [ ] Verify Slack notifications working - [ ] Check VPN/access working - [ ] Review critical dashboards ```` ### Template 2: Quick Handoff (Async) ```markdown # Quick Handoff: @alice β†’ @bob ## TL;DR - No active incidents - 1 investigation ongoing (API timeouts, see ENG-1234) - Major release tomorrow (01/24) - be ready for issues ## Watch List 1. API latency around 02:00-03:00 UTC (backup window) 2. Auth service memory (restart if > 80%) ## Recent - Deployed api-gateway v3.2.1 yesterday (stable) - Increased rate limits to 1500 RPS ## Coming Up - 01/23 02:00 - DB maintenance (5 min read-only) - 01/24 14:00 - v5.0 release ## Questions? I'll be available on Slack until 17:00 today. ```` ### Template 3: Incident Handoff (Mid-Incident) ```markdown # INCIDENT HANDOFF: Payment Service Degradation **Incident Start**: 2024-01-22 08:15 UTC **Current Status**: Mitigating **Severity**: SEV2 --- ## Current State - Error rate: 15% (down from 40%) - Mitigation in progress: scaling up pods - ETA to resolution: ~30 min ## What We Know 1. Root cause: Memory pressure on payment-service pods 2. Triggered by: Unusual traffic spike (3x normal) 3. Contributing: Inefficient query in checkout flow ## What We've Done - Scaled payment-service from 5 β†’ 15 pods - Enabled rate limiting on checkout endpoint - Disabled non-critical features ## What Needs to Happen 1. Monitor error rate - should reach <1% in ~15 min 2. If not improving, escalate to @payments-manager 3. Once stable, begin root cause investigation ## Key People - Incident Commander: @alice (handing off) - Comms Lead: @charlie - Technical Lead: @bob (incoming) ## Communication - Status page: Updated at 08:45 - Customer support: Notified - Exec team: Aware ## Resources - Incident channel: #inc-20240122-payment - Dashboard: [Payment Service](https://grafana/d/payments) - Runbook: [Payment Degradation](https://wiki/runbooks/payments) --- **Incoming on-call (@bob) - Please confirm you have:** - [ ] Joined #inc-20240122-payment - [ ] Access to dashboards - [ ] Understand current state - [ ] Know escalation path ``` ## Handoff Sync Meeting ### Agenda (15 minutes) ```markdown ## Handoff Sync: @alice β†’ @bob 1. **Active Issues** (5 min) - Walk through any ongoing incidents - Discuss investigation status - Transfer context and theories 2. **Recent Changes** (3 min) - Deployments to watch - Config changes - Known regressions 3. **Upcoming Events** (3 min) - Maintenance windows - Expected traffic changes - Releases planned 4. **Questions** (4 min) - Clarify anything unclear - Confirm access and alerting - Exchange contact info ``` ## On-Call Best Practices ### Before Your Shift ```markdown ## Pre-Shift Checklist ### Access Verification - [ ] VPN working - [ ] kubectl access to all clusters - [ ] Database read access - [ ] Log aggregator access (Splunk/Datadog) - [ ] PagerDuty app installed and logged in ### Alerting Setup - [ ] PagerDuty schedule shows you as primary - [ ] Phone notifications enabled - [ ] Slack notifications for incident channels - [ ] Test alert received and acknowledged ### Knowledge Refresh - [ ] Review recent incidents (past 2 weeks) - [ ] Check service changelog - [ ] Skim critical runbooks - [ ] Know escalation contacts ### Environment Ready - [ ] Laptop charged and accessible - [ ] Phone charged - [ ] Quiet space available for calls - [ ] Secondary contact identified (if traveling) ``` ### During Your Shift ```markdown ## Daily On-Call Routine ### Morning (start of day) - [ ] Check overnight alerts - [ ] Review dashboards for anomalies - [ ] Check for any P0/P1 tickets created - [ ] Skim incident channels for context ### Throughout Day - [ ] Respond to alerts within SLA - [ ] Document investigation progress - [ ] Update team on significant issues - [ ] Triage incoming pages ### End of Day - [ ] Hand off any active issues - [ ] Update investigation docs - [ ] Note anything for next shift ``` ### After Your Shift ```markdown ## Post-Shift Checklist - [ ] Complete handoff document - [ ] Sync with incoming on-call - [ ] Verify PagerDuty routing changed - [ ] Close/update investigation tickets - [ ] File postmortems for any incidents - [ ] Take time off if shift was stressful ``` ## Escalation Guidelines ### When to Escalate ```markdown ## Escalation Triggers ### Immediate Escalation - SEV1 incident declared - Data breach suspected - Unable to diagnose within 30 min - Customer or legal escalation received ### Consider Escalation - Issue spans multiple teams - Requires expertise you don't have - Business impact exceeds threshold - You're uncertain about next steps ### How to Escalate 1. Page the appropriate escalation path 2. Provide brief context in Slack 3. Stay engaged until escalation acknowledges 4. Hand off cleanly, don't just disappear ``` ## Best Practices ### Do's - **Document everything** - Future you will thank you - **Escalate early** - Better safe than sorry - **Take breaks** - Alert fatigue is real - **Keep handoffs synchronous** - Async loses context - **Test your setup** - Before incidents, not during ### Don'ts - **Don't skip handoffs** - Context loss causes incidents - **Don't hero** - Escalate when needed - **Don't ignore alerts** - Even if they seem minor - **Don't work sick** - Swap shifts instead - **Don't disappear** - Stay reachable during shift ## Resources - [Google SRE - Being On-Call](https://sre.google/sre-book/being-on-call/) - [PagerDuty On-Call Guide](https://www.pagerduty.com/resources/learn/on-call-management/) - [Increment On-Call Issue](https://increment.com/on-call/)
πŸ‘0
πŸ‘οΈ0
πŸ€– Auto-discovered
πŸ€–system promptβ€’7 months ago

postmortem-writing

Write effective blameless postmortems with root cause analysis,

coding
⭐1
# Postmortem Writing Comprehensive guide to writing effective, blameless postmortems that drive organizational learning and prevent incident recurrence. ## When to Use This Skill - Conducting post-incident reviews - Writing postmortem documents - Facilitating blameless postmortem meetings - Identifying root causes and contributing factors - Creating actionable follow-up items - Building organizational learning culture ## Core Concepts ### 1. Blameless Culture | Blame-Focused | Blameless | | ------------------------ | --------------------------------- | | "Who caused this?" | "What conditions allowed this?" | | "Someone made a mistake" | "The system allowed this mistake" | | Punish individuals | Improve systems | | Hide information | Share learnings | | Fear of speaking up | Psychological safety | ### 2. Postmortem Triggers - SEV1 or SEV2 incidents - Customer-facing outages > 15 minutes - Data loss or security incidents - Near-misses that could have been severe - Novel failure modes - Incidents requiring unusual intervention ## Quick Start ### Postmortem Timeline ``` Day 0: Incident occurs Day 1-2: Draft postmortem document Day 3-5: Postmortem meeting Day 5-7: Finalize document, create tickets Week 2+: Action item completion Quarterly: Review patterns across incidents ``` ## Templates ### Template 1: Standard Postmortem ```markdown # Postmortem: [Incident Title] **Date**: 2024-01-15 **Authors**: @alice, @bob **Status**: Draft | In Review | Final **Incident Severity**: SEV2 **Incident Duration**: 47 minutes ## Executive Summary On January 15, 2024, the payment processing service experienced a 47-minute outage affecting approximately 12,000 customers. The root cause was a database connection pool exhaustion triggered by a configuration change in deployment v2.3.4. The incident was resolved by rolling back to v2.3.3 and increasing connection pool limits. **Impact**: - 12,000 customers unable to complete purchases - Estimated revenue loss: $45,000 - 847 support tickets created - No data loss or security implications ## Timeline (All times UTC) | Time | Event | | ----- | ----------------------------------------------- | | 14:23 | Deployment v2.3.4 completed to production | | 14:31 | First alert: `payment_error_rate > 5%` | | 14:33 | On-call engineer @alice acknowledges alert | | 14:35 | Initial investigation begins, error rate at 23% | | 14:41 | Incident declared SEV2, @bob joins | | 14:45 | Database connection exhaustion identified | | 14:52 | Decision to rollback deployment | | 14:58 | Rollback to v2.3.3 initiated | | 15:10 | Rollback complete, error rate dropping | | 15:18 | Service fully recovered, incident resolved | ## Root Cause Analysis ### What Happened The v2.3.4 deployment included a change to the database query pattern that inadvertently removed connection pooling for a frequently-called endpoint. Each request opened a new database connection instead of reusing pooled connections. ### Why It Happened 1. **Proximate Cause**: Code change in `PaymentRepository.java` replaced pooled `DataSource` with direct `DriverManager.getConnection()` calls. 2. **Contributing Factors**: - Code review did not catch the connection handling change - No integration tests specifically for connection pool behavior - Staging environment has lower traffic, masking the issue - Database connection metrics alert threshold was too high (90%) 3. **5 Whys Analysis**: - Why did the service fail? β†’ Database connections exhausted - Why were connections exhausted? β†’ Each request opened new connection - Why did each request open new connection? β†’ Code bypassed connection pool - Why did code bypass connection pool? β†’ Developer unfamiliar with codebase patterns - Why was developer unfamiliar? β†’ No documentation on connection management patterns ### System Diagram ``` [Client] β†’ [Load Balancer] β†’ [Payment Service] β†’ [Database] ↓ Connection Pool (broken) ↓ Direct connections (cause) ``` ## Detection ### What Worked - Error rate alert fired within 8 minutes of deployment - Grafana dashboard clearly showed connection spike - On-call response was swift (2 minute acknowledgment) ### What Didn't Work - Database connection metric alert threshold too high - No deployment-correlated alerting - Canary deployment would have caught this earlier ### Detection Gap The deployment completed at 14:23, but the first alert didn't fire until 14:31 (8 minutes). A deployment-aware alert could have detected the issue faster. ## Response ### What Worked - On-call engineer quickly identified database as the issue - Rollback decision was made decisively - Clear communication in incident channel ### What Could Be Improved - Took 10 minutes to correlate issue with recent deployment - Had to manually check deployment history - Rollback took 12 minutes (could be faster) ## Impact ### Customer Impact - 12,000 unique customers affected - Average impact duration: 35 minutes - 847 support tickets (23% of affected users) - Customer satisfaction score dropped 12 points ### Business Impact - Estimated revenue loss: $45,000 - Support cost: ~$2,500 (agent time) - Engineering time: ~8 person-hours ### Technical Impact - Database primary experienced elevated load - Some replica lag during incident - No permanent damage to systems ## Lessons Learned ### What Went Well 1. Alerting detected the issue before customer reports 2. Team collaborated effectively under pressure 3. Rollback procedure worked smoothly 4. Communication was clear and timely ### What Went Wrong 1. Code review missed critical change 2. Test coverage gap for connection pooling 3. Staging environment doesn't reflect production traffic 4. Alert thresholds were not tuned properly ### Where We Got Lucky 1. Incident occurred during business hours with full team available 2. Database handled the load without failing completely 3. No other incidents occurred simultaneously ## Action Items | Priority | Action | Owner | Due Date | Ticket | |----------|--------|-------|----------|--------| | P0 | Add integration test for connection pool behavior | @alice | 2024-01-22 | ENG-1234 | | P0 | Lower database connection alert threshold to 70% | @bob | 2024-01-17 | OPS-567 | | P1 | Document connection management patterns | @alice | 2024-01-29 | DOC-89 | | P1 | Implement deployment-correlated alerting | @bob | 2024-02-05 | OPS-568 | | P2 | Evaluate canary deployment strategy | @charlie | 2024-02-15 | ENG-1235 | | P2 | Load test staging with production-like traffic | @dave | 2024-02-28 | QA-123 | ## Appendix ### Supporting Data #### Error Rate Graph [Link to Grafana dashboard snapshot] #### Database Connection Graph [Link to metrics] ### Related Incidents - 2023-11-02: Similar connection issue in User Service (POSTMORTEM-42) ### References - [Connection Pool Best Practices](internal-wiki/connection-pools) - [Deployment Runbook](internal-wiki/deployment-runbook) ``` ### Template 2: 5 Whys Analysis ```markdown # 5 Whys Analysis: [Incident] ## Problem Statement Payment service experienced 47-minute outage due to database connection exhaustion. ## Analysis ### Why #1: Why did the service fail? **Answer**: Database connections were exhausted, causing all new requests to fail. **Evidence**: Metrics showed connection count at 100/100 (max), with 500+ pending requests. --- ### Why #2: Why were database connections exhausted? **Answer**: Each incoming request opened a new database connection instead of using the connection pool. **Evidence**: Code diff shows direct `DriverManager.getConnection()` instead of pooled `DataSource`. --- ### Why #3: Why did the code bypass the connection pool? **Answer**: A developer refactored the repository class and inadvertently changed the connection acquisition method. **Evidence**: PR #1234 shows the change, made while fixing a different bug. --- ### Why #4: Why wasn't this caught in code review? **Answer**: The reviewer focused on the functional change (the bug fix) and didn't notice the infrastructure change. **Evidence**: Review comments only discuss business logic. --- ### Why #5: Why isn't there a safety net for this type of change? **Answer**: We lack automated tests that verify connection pool behavior and lack documentation about our connection patterns. **Evidence**: Test suite has no tests for connection handling; wiki has no article on database connections. ## Root Causes Identified 1. **Primary**: Missing automated tests for infrastructure behavior 2. **Secondary**: Insufficient documentation of architectural patterns 3. **Tertiary**: Code review checklist doesn't include infrastructure considerations ## Systemic Improvements | Root Cause | Improvement | Type | | ------------- | --------------------------------- | ---------- | | Missing tests | Add infrastructure behavior tests | Prevention | | Missing docs | Document connection patterns | Prevention | | Review gaps | Update review checklist | Detection | | No canary | Implement canary deployments | Mitigation | ``` ### Template 3: Quick Postmortem (Minor Incidents) ```markdown # Quick Postmortem: [Brief Title] **Date**: 2024-01-15 | **Duration**: 12 min | **Severity**: SEV3 ## What Happened API latency spiked to 5s due to cache miss storm after cache flush. ## Timeline - 10:00 - Cache flush initiated for config update - 10:02 - Latency alerts fire - 10:05 - Identified as cache miss storm - 10:08 - Enabled cache warming - 10:12 - Latency normalized ## Root Cause Full cache flush for minor config update caused thundering herd. ## Fix - Immediate: Enabled cache warming - Long-term: Implement partial cache invalidation (ENG-999) ## Lessons Don't full-flush cache in production; use targeted invalidation. ``` ## Facilitation Guide ### Running a Postmortem Meeting ```markdown ## Meeting Structure (60 minutes) ### 1. Opening (5 min) - Remind everyone of blameless culture - "We're here to learn, not to blame" - Review meeting norms ### 2. Timeline Review (15 min) - Walk through events chronologically - Ask clarifying questions - Identify gaps in timeline ### 3. Analysis Discussion (20 min) - What failed? - Why did it fail? - What conditions allowed this? - What would have prevented it? ### 4. Action Items (15 min) - Brainstorm improvements - Prioritize by impact and effort - Assign owners and due dates ### 5. Closing (5 min) - Summarize key learnings - Confirm action item owners - Schedule follow-up if needed ## Facilitation Tips - Keep discussion on track - Redirect blame to systems - Encourage quiet participants - Document dissenting views - Time-box tangents ``` ## Anti-Patterns to Avoid | Anti-Pattern | Problem | Better Approach | | ----------------------- | -------------------------- | ------------------------------- | | **Blame game** | Shuts down learning | Focus on systems | | **Shallow analysis** | Doesn't prevent recurrence | Ask "why" 5 times | | **No action items** | Waste of time | Always have concrete next steps | | **Unrealistic actions** | Never completed | Scope to achievable tasks | | **No follow-up** | Actions forgotten | Track in ticketing system | ## Best Practices ### Do's - **Start immediately** - Memory fades fast - **Be specific** - Exact times, exact errors - **Include graphs** - Visual evidence - **Assign owners** - No orphan action items - **Share widely** - Organizational learning ### Don'ts - **Don't name and shame** - Ever - **Don't skip small incidents** - They reveal patterns - **Don't make it a blame doc** - That kills learning - **Don't create busywork** - Actions should be meaningful - **Don't skip follow-up** - Verify actions completed ## Resources - [Google SRE - Postmortem Culture](https://sre.google/sre-book/postmortem-culture/) - [Etsy's Blameless Postmortems](https://codeascraft.com/2012/05/22/blameless-postmortems/) - [PagerDuty Postmortem Guide](https://postmortems.pagerduty.com/)
πŸ‘0
πŸ‘οΈ0
πŸ€– Auto-discovered
πŸ€–system promptβ€’7 months ago

backtesting-frameworks

Build robust backtesting systems for trading strategies with proper

coding
⭐1
# Backtesting Frameworks Build robust, production-grade backtesting systems that avoid common pitfalls and produce reliable strategy performance estimates. ## When to Use This Skill - Developing trading strategy backtests - Building backtesting infrastructure - Validating strategy performance - Avoiding common backtesting biases - Implementing walk-forward analysis - Comparing strategy alternatives ## Core Concepts ### 1. Backtesting Biases | Bias | Description | Mitigation | | ---------------- | ------------------------- | ----------------------- | | **Look-ahead** | Using future information | Point-in-time data | | **Survivorship** | Only testing on survivors | Use delisted securities | | **Overfitting** | Curve-fitting to history | Out-of-sample testing | | **Selection** | Cherry-picking strategies | Pre-registration | | **Transaction** | Ignoring trading costs | Realistic cost models | ### 2. Proper Backtest Structure ``` Historical Data β”‚ β–Ό β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” β”‚ Training Set β”‚ β”‚ (Strategy Development & Optimization) β”‚ β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜ β”‚ β–Ό β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” β”‚ Validation Set β”‚ β”‚ (Parameter Selection, No Peeking) β”‚ β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜ β”‚ β–Ό β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” β”‚ Test Set β”‚ β”‚ (Final Performance Evaluation) β”‚ β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜ ``` ### 3. Walk-Forward Analysis ``` Window 1: [Train──────][Test] Window 2: [Train──────][Test] Window 3: [Train──────][Test] Window 4: [Train──────][Test] ─────▢ Time ``` ## Implementation Patterns ### Pattern 1: Event-Driven Backtester ```python from abc import ABC, abstractmethod from dataclasses import dataclass, field from datetime import datetime from decimal import Decimal from enum import Enum from typing import Dict, List, Optional import pandas as pd import numpy as np class OrderSide(Enum): BUY = "buy" SELL = "sell" class OrderType(Enum): MARKET = "market" LIMIT = "limit" STOP = "stop" @dataclass class Order: symbol: str side: OrderSide quantity: Decimal order_type: OrderType limit_price: Optional[Decimal] = None stop_price: Optional[Decimal] = None timestamp: Optional[datetime] = None @dataclass class Fill: order: Order fill_price: Decimal fill_quantity: Decimal commission: Decimal slippage: Decimal timestamp: datetime @dataclass class Position: symbol: str quantity: Decimal = Decimal("0") avg_cost: Decimal = Decimal("0") realized_pnl: Decimal = Decimal("0") def update(self, fill: Fill) -> None: if fill.order.side == OrderSide.BUY: new_quantity = self.quantity + fill.fill_quantity if new_quantity != 0: self.avg_cost = ( (self.quantity * self.avg_cost + fill.fill_quantity * fill.fill_price) / new_quantity ) self.quantity = new_quantity else: self.realized_pnl += fill.fill_quantity * (fill.fill_price - self.avg_cost) self.quantity -= fill.fill_quantity @dataclass class Portfolio: cash: Decimal positions: Dict[str, Position] = field(default_factory=dict) def get_position(self, symbol: str) -> Position: if symbol not in self.positions: self.positions[symbol] = Position(symbol=symbol) return self.positions[symbol] def process_fill(self, fill: Fill) -> None: position = self.get_position(fill.order.symbol) position.update(fill) if fill.order.side == OrderSide.BUY: self.cash -= fill.fill_price * fill.fill_quantity + fill.commission else: self.cash += fill.fill_price * fill.fill_quantity - fill.commission def get_equity(self, prices: Dict[str, Decimal]) -> Decimal: equity = self.cash for symbol, position in self.positions.items(): if position.quantity != 0 and symbol in prices: equity += position.quantity * prices[symbol] return equity class Strategy(ABC): @abstractmethod def on_bar(self, timestamp: datetime, data: pd.DataFrame) -> List[Order]: pass @abstractmethod def on_fill(self, fill: Fill) -> None: pass class ExecutionModel(ABC): @abstractmethod def execute(self, order: Order, bar: pd.Series) -> Optional[Fill]: pass class SimpleExecutionModel(ExecutionModel): def __init__(self, slippage_bps: float = 10, commission_per_share: float = 0.01): self.slippage_bps = slippage_bps self.commission_per_share = commission_per_share def execute(self, order: Order, bar: pd.Series) -> Optional[Fill]: if order.order_type == OrderType.MARKET: base_price = Decimal(str(bar["open"])) # Apply slippage slippage_mult = 1 + (self.slippage_bps / 10000) if order.side == OrderSide.BUY: fill_price = base_price * Decimal(str(slippage_mult)) else: fill_price = base_price / Decimal(str(slippage_mult)) commission = order.quantity * Decimal(str(self.commission_per_share)) slippage = abs(fill_price - base_price) * order.quantity return Fill( order=order, fill_price=fill_price, fill_quantity=order.quantity, commission=commission, slippage=slippage, timestamp=bar.name ) return None class Backtester: def __init__( self, strategy: Strategy, execution_model: ExecutionModel, initial_capital: Decimal = Decimal("100000") ): self.strategy = strategy self.execution_model = execution_model self.portfolio = Portfolio(cash=initial_capital) self.equity_curve: List[tuple] = [] self.trades: List[Fill] = [] def run(self, data: pd.DataFrame) -> pd.DataFrame: """Run backtest on OHLCV data with DatetimeIndex.""" pending_orders: List[Order] = [] for timestamp, bar in data.iterrows(): # Execute pending orders at today's prices for order in pending_orders: fill = self.execution_model.execute(order, bar) if fill: self.portfolio.process_fill(fill) self.strategy.on_fill(fill) self.trades.append(fill) pending_orders.clear() # Get current prices for equity calculation prices = {data.index.name or "default": Decimal(str(bar["close"]))} equity = self.portfolio.get_equity(prices) self.equity_curve.append((timestamp, float(equity))) # Generate new orders for next bar new_orders = self.strategy.on_bar(timestamp, data.loc[:timestamp]) pending_orders.extend(new_orders) return self._create_results() def _create_results(self) -> pd.DataFrame: equity_df = pd.DataFrame(self.equity_curve, columns=["timestamp", "equity"]) equity_df.set_index("timestamp", inplace=True) equity_df["returns"] = equity_df["equity"].pct_change() return equity_df ``` ### Pattern 2: Vectorized Backtester (Fast) ```python import pandas as pd import numpy as np from typing import Callable, Dict, Any class VectorizedBacktester: """Fast vectorized backtester for simple strategies.""" def __init__( self, initial_capital: float = 100000, commission: float = 0.001, # 0.1% slippage: float = 0.0005 # 0.05% ): self.initial_capital = initial_capital self.commission = commission self.slippage = slippage def run( self, prices: pd.DataFrame, signal_func: Callable[[pd.DataFrame], pd.Series] ) -> Dict[str, Any]: """ Run backtest with signal function. Args: prices: DataFrame with 'close' column signal_func: Function that returns position signals (-1, 0, 1) Returns: Dictionary with results """ # Generate signals (shifted to avoid look-ahead) signals = signal_func(prices).shift(1).fillna(0) # Calculate returns returns = prices["close"].pct_change() # Calculate strategy returns with costs position_changes = signals.diff().abs() trading_costs = position_changes * (self.commission + self.slippage) strategy_returns = signals * returns - trading_costs # Build equity curve equity = (1 + strategy_returns).cumprod() * self.initial_capital # Calculate metrics results = { "equity": equity, "returns": strategy_returns, "signals": signals, "metrics": self._calculate_metrics(strategy_returns, equity) } return results def _calculate_metrics( self, returns: pd.Series, equity: pd.Series ) -> Dict[str, float]: """Calculate performance metrics.""" total_return = (equity.iloc[-1] / self.initial_capital) - 1 annual_return = (1 + total_return) ** (252 / len(returns)) - 1 annual_vol = returns.std() * np.sqrt(252) sharpe = annual_return / annual_vol if annual_vol > 0 else 0 # Drawdown rolling_max = equity.cummax() drawdown = (equity - rolling_max) / rolling_max max_drawdown = drawdown.min() # Win rate winning_days = (returns > 0).sum() total_days = (returns != 0).sum() win_rate = winning_days / total_days if total_days > 0 else 0 return { "total_return": total_return, "annual_return": annual_return, "annual_volatility": annual_vol, "sharpe_ratio": sharpe, "max_drawdown": max_drawdown, "win_rate": win_rate, "num_trades": int((returns != 0).sum()) } # Example usage def momentum_signal(prices: pd.DataFrame, lookback: int = 20) -> pd.Series: """Simple momentum strategy: long when price > SMA, else flat.""" sma = prices["close"].rolling(lookback).mean() return (prices["close"] > sma).astype(int) # Run backtest # backtester = VectorizedBacktester() # results = backtester.run(price_data, lambda p: momentum_signal(p, 50)) ``` ### Pattern 3: Walk-Forward Optimization ```python from typing import Callable, Dict, List, Tuple, Any import pandas as pd import numpy as np from itertools import product class WalkForwardOptimizer: """Walk-forward analysis with anchored or rolling windows.""" def __init__( self, train_period: int, test_period: int, anchored: bool = False, n_splits: int = None ): """ Args: train_period: Number of bars in training window test_period: Number of bars in test window anchored: If True, training always starts from beginning n_splits: Number of train/test splits (auto-calculated if None) """ self.train_period = train_period self.test_period = test_period self.anchored = anchored self.n_splits = n_splits def generate_splits( self, data: pd.DataFrame ) -> List[Tuple[pd.DataFrame, pd.DataFrame]]: """Generate train/test splits.""" splits = [] n = len(data) if self.n_splits: step = (n - self.train_period) // self.n_splits else: step = self.test_period start = 0 while start + self.train_period + self.test_period <= n: if self.anchored: train_start = 0 else: train_start = start train_end = start + self.train_period test_end = min(train_end + self.test_period, n) train_data = data.iloc[train_start:train_end] test_data = data.iloc[train_end:test_end] splits.append((train_data, test_data)) start += step return splits def optimize( self, data: pd.DataFrame, strategy_func: Callable, param_grid: Dict[str, List], metric: str = "sharpe_ratio" ) -> Dict[str, Any]: """ Run walk-forward optimization. Args: data: Full dataset strategy_func: Function(data, **params) -> results dict param_grid: Parameter combinations to test metric: Metric to optimize Returns: Combined results from all test periods """ splits = self.generate_splits(data) all_results = [] optimal_params_history = [] for i, (train_data, test_data) in enumerate(splits): # Optimize on training data best_params, best_metric = self._grid_search( train_data, strategy_func, param_grid, metric ) optimal_params_history.append(best_params) # Test with optimal params test_results = strategy_func(test_data, **best_params) test_results["split"] = i test_results["params"] = best_params all_results.append(test_results) print(f"Split {i+1}/{len(splits)}: " f"Best {metric}={best_metric:.4f}, params={best_params}") return { "split_results": all_results, "param_history": optimal_params_history, "combined_equity": self._combine_equity_curves(all_results) } def _grid_search( self, data: pd.DataFrame, strategy_func: Callable, param_grid: Dict[str, List], metric: str ) -> Tuple[Dict, float]: """Grid search for best parameters.""" best_params = None best_metric = -np.inf # Generate all parameter combinations param_names = list(param_grid.keys()) param_values = list(param_grid.values()) for values in product(*param_values): params = dict(zip(param_names, values)) results = strategy_func(data, **params) if results["metrics"][metric] > best_metric: best_metric = results["metrics"][metric] best_params = params return best_params, best_metric def _combine_equity_curves( self, results: List[Dict] ) -> pd.Series: """Combine equity curves from all test periods.""" combined = pd.concat([r["equity"] for r in results]) return combined ``` ### Pattern 4: Monte Carlo Analysis ```python import numpy as np import pandas as pd from typing import Dict, List class MonteCarloAnalyzer: """Monte Carlo simulation for strategy robustness.""" def __init__(self, n_simulations: int = 1000, confidence: float = 0.95): self.n_simulations = n_simulations self.confidence = confidence def bootstrap_returns( self, returns: pd.Series, n_periods: int = None ) -> np.ndarray: """ Bootstrap simulation by resampling returns. Args: returns: Historical returns series n_periods: Length of each simulation (default: same as input) Returns: Array of shape (n_simulations, n_periods) """ if n_periods is None: n_periods = len(returns) simulations = np.zeros((self.n_simulations, n_periods)) for i in range(self.n_simulations): # Resample with replacement simulated_returns = np.random.choice( returns.values, size=n_periods, replace=True ) simulations[i] = simulated_returns return simulations def analyze_drawdowns( self, returns: pd.Series ) -> Dict[str, float]: """Analyze drawdown distribution via simulation.""" simulations = self.bootstrap_returns(returns) max_drawdowns = [] for sim_returns in simulations: equity = (1 + sim_returns).cumprod() rolling_max = np.maximum.accumulate(equity) drawdowns = (equity - rolling_max) / rolling_max max_drawdowns.append(drawdowns.min()) max_drawdowns = np.array(max_drawdowns) return { "expected_max_dd": np.mean(max_drawdowns), "median_max_dd": np.median(max_drawdowns), f"worst_{int(self.confidence*100)}pct": np.percentile( max_drawdowns, (1 - self.confidence) * 100 ), "worst_case": max_drawdowns.min() } def probability_of_loss( self, returns: pd.Series, holding_periods: List[int] = [21, 63, 126, 252] ) -> Dict[int, float]: """Calculate probability of loss over various holding periods.""" results = {} for period in holding_periods: if period > len(returns): continue simulations = self.bootstrap_returns(returns, period) total_returns = (1 + simulations).prod(axis=1) - 1 prob_loss = (total_returns < 0).mean() results[period] = prob_loss return results def confidence_interval( self, returns: pd.Series, periods: int = 252 ) -> Dict[str, float]: """Calculate confidence interval for future returns.""" simulations = self.bootstrap_returns(returns, periods) total_returns = (1 + simulations).prod(axis=1) - 1 lower = (1 - self.confidence) / 2 upper = 1 - lower return { "expected": total_returns.mean(), "lower_bound": np.percentile(total_returns, lower * 100), "upper_bound": np.percentile(total_returns, upper * 100), "std": total_returns.std() } ``` ## Performance Metrics ```python def calculate_metrics(returns: pd.Series, rf_rate: float = 0.02) -> Dict[str, float]: """Calculate comprehensive performance metrics.""" # Annualization factor (assuming daily returns) ann_factor = 252 # Basic metrics total_return = (1 + returns).prod() - 1 annual_return = (1 + total_return) ** (ann_factor / len(returns)) - 1 annual_vol = returns.std() * np.sqrt(ann_factor) # Risk-adjusted returns sharpe = (annual_return - rf_rate) / annual_vol if annual_vol > 0 else 0 # Sortino (downside deviation) downside_returns = returns[returns < 0] downside_vol = downside_returns.std() * np.sqrt(ann_factor) sortino = (annual_return - rf_rate) / downside_vol if downside_vol > 0 else 0 # Calmar ratio equity = (1 + returns).cumprod() rolling_max = equity.cummax() drawdowns = (equity - rolling_max) / rolling_max max_drawdown = drawdowns.min() calmar = annual_return / abs(max_drawdown) if max_drawdown != 0 else 0 # Win rate and profit factor wins = returns[returns > 0] losses = return
πŸ‘0
πŸ‘οΈ0
πŸ€– Auto-discovered
πŸ€–system promptβ€’7 months ago

attack-tree-construction

Build comprehensive attack trees to visualize threat paths. Use

security
⭐1
# Attack Tree Construction Systematic attack path visualization and analysis. ## When to Use This Skill - Visualizing complex attack scenarios - Identifying defense gaps and priorities - Communicating risks to stakeholders - Planning defensive investments - Penetration test planning - Security architecture review ## Core Concepts ### 1. Attack Tree Structure ``` [Root Goal] | β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”΄β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” β”‚ β”‚ [Sub-goal 1] [Sub-goal 2] (OR node) (AND node) β”‚ β”‚ β”Œβ”€β”€β”€β”€β”€β”΄β”€β”€β”€β”€β”€β” β”Œβ”€β”€β”€β”€β”€β”΄β”€β”€β”€β”€β”€β” β”‚ β”‚ β”‚ β”‚ [Attack] [Attack] [Attack] [Attack] (leaf) (leaf) (leaf) (leaf) ``` ### 2. Node Types | Type | Symbol | Description | | -------- | --------- | ----------------------- | | **OR** | Oval | Any child achieves goal | | **AND** | Rectangle | All children required | | **Leaf** | Box | Atomic attack step | ### 3. Attack Attributes | Attribute | Description | Values | | ------------- | ----------------------- | ------------------ | | **Cost** | Resources needed | $, $$, $$$ | | **Time** | Duration to execute | Hours, Days, Weeks | | **Skill** | Expertise required | Low, Medium, High | | **Detection** | Likelihood of detection | Low, Medium, High | ## Templates ### Template 1: Attack Tree Data Model ```python from dataclasses import dataclass, field from enum import Enum from typing import List, Dict, Optional, Union import json class NodeType(Enum): OR = "or" AND = "and" LEAF = "leaf" class Difficulty(Enum): TRIVIAL = 1 LOW = 2 MEDIUM = 3 HIGH = 4 EXPERT = 5 class Cost(Enum): FREE = 0 LOW = 1 MEDIUM = 2 HIGH = 3 VERY_HIGH = 4 class DetectionRisk(Enum): NONE = 0 LOW = 1 MEDIUM = 2 HIGH = 3 CERTAIN = 4 @dataclass class AttackAttributes: difficulty: Difficulty = Difficulty.MEDIUM cost: Cost = Cost.MEDIUM detection_risk: DetectionRisk = DetectionRisk.MEDIUM time_hours: float = 8.0 requires_insider: bool = False requires_physical: bool = False @dataclass class AttackNode: id: str name: str description: str node_type: NodeType attributes: AttackAttributes = field(default_factory=AttackAttributes) children: List['AttackNode'] = field(default_factory=list) mitigations: List[str] = field(default_factory=list) cve_refs: List[str] = field(default_factory=list) def add_child(self, child: 'AttackNode') -> None: self.children.append(child) def calculate_path_difficulty(self) -> float: """Calculate aggregate difficulty for this path.""" if self.node_type == NodeType.LEAF: return self.attributes.difficulty.value if not self.children: return 0 child_difficulties = [c.calculate_path_difficulty() for c in self.children] if self.node_type == NodeType.OR: return min(child_difficulties) else: # AND return max(child_difficulties) def calculate_path_cost(self) -> float: """Calculate aggregate cost for this path.""" if self.node_type == NodeType.LEAF: return self.attributes.cost.value if not self.children: return 0 child_costs = [c.calculate_path_cost() for c in self.children] if self.node_type == NodeType.OR: return min(child_costs) else: # AND return sum(child_costs) def to_dict(self) -> Dict: """Convert to dictionary for serialization.""" return { "id": self.id, "name": self.name, "description": self.description, "type": self.node_type.value, "attributes": { "difficulty": self.attributes.difficulty.name, "cost": self.attributes.cost.name, "detection_risk": self.attributes.detection_risk.name, "time_hours": self.attributes.time_hours, }, "mitigations": self.mitigations, "children": [c.to_dict() for c in self.children] } @dataclass class AttackTree: name: str description: str root: AttackNode version: str = "1.0" def find_easiest_path(self) -> List[AttackNode]: """Find the path with lowest difficulty.""" return self._find_path(self.root, minimize="difficulty") def find_cheapest_path(self) -> List[AttackNode]: """Find the path with lowest cost.""" return self._find_path(self.root, minimize="cost") def find_stealthiest_path(self) -> List[AttackNode]: """Find the path with lowest detection risk.""" return self._find_path(self.root, minimize="detection") def _find_path( self, node: AttackNode, minimize: str ) -> List[AttackNode]: """Recursive path finding.""" if node.node_type == NodeType.LEAF: return [node] if not node.children: return [node] if node.node_type == NodeType.OR: # Pick the best child path best_path = None best_score = float('inf') for child in node.children: child_path = self._find_path(child, minimize) score = self._path_score(child_path, minimize) if score < best_score: best_score = score best_path = child_path return [node] + (best_path or []) else: # AND # Must traverse all children path = [node] for child in node.children: path.extend(self._find_path(child, minimize)) return path def _path_score(self, path: List[AttackNode], metric: str) -> float: """Calculate score for a path.""" if metric == "difficulty": return sum(n.attributes.difficulty.value for n in path if n.node_type == NodeType.LEAF) elif metric == "cost": return sum(n.attributes.cost.value for n in path if n.node_type == NodeType.LEAF) elif metric == "detection": return sum(n.attributes.detection_risk.value for n in path if n.node_type == NodeType.LEAF) return 0 def get_all_leaf_attacks(self) -> List[AttackNode]: """Get all leaf attack nodes.""" leaves = [] self._collect_leaves(self.root, leaves) return leaves def _collect_leaves(self, node: AttackNode, leaves: List[AttackNode]) -> None: if node.node_type == NodeType.LEAF: leaves.append(node) for child in node.children: self._collect_leaves(child, leaves) def get_unmitigated_attacks(self) -> List[AttackNode]: """Find attacks without mitigations.""" return [n for n in self.get_all_leaf_attacks() if not n.mitigations] def export_json(self) -> str: """Export tree to JSON.""" return json.dumps({ "name": self.name, "description": self.description, "version": self.version, "root": self.root.to_dict() }, indent=2) ``` ### Template 2: Attack Tree Builder ```python class AttackTreeBuilder: """Fluent builder for attack trees.""" def __init__(self, name: str, description: str): self.name = name self.description = description self._node_stack: List[AttackNode] = [] self._root: Optional[AttackNode] = None def goal(self, id: str, name: str, description: str = "") -> 'AttackTreeBuilder': """Set the root goal (OR node by default).""" self._root = AttackNode( id=id, name=name, description=description, node_type=NodeType.OR ) self._node_stack = [self._root] return self def or_node(self, id: str, name: str, description: str = "") -> 'AttackTreeBuilder': """Add an OR sub-goal.""" node = AttackNode( id=id, name=name, description=description, node_type=NodeType.OR ) self._current().add_child(node) self._node_stack.append(node) return self def and_node(self, id: str, name: str, description: str = "") -> 'AttackTreeBuilder': """Add an AND sub-goal (all children required).""" node = AttackNode( id=id, name=name, description=description, node_type=NodeType.AND ) self._current().add_child(node) self._node_stack.append(node) return self def attack( self, id: str, name: str, description: str = "", difficulty: Difficulty = Difficulty.MEDIUM, cost: Cost = Cost.MEDIUM, detection: DetectionRisk = DetectionRisk.MEDIUM, time_hours: float = 8.0, mitigations: List[str] = None ) -> 'AttackTreeBuilder': """Add a leaf attack node.""" node = AttackNode( id=id, name=name, description=description, node_type=NodeType.LEAF, attributes=AttackAttributes( difficulty=difficulty, cost=cost, detection_risk=detection, time_hours=time_hours ), mitigations=mitigations or [] ) self._current().add_child(node) return self def end(self) -> 'AttackTreeBuilder': """Close current node, return to parent.""" if len(self._node_stack) > 1: self._node_stack.pop() return self def build(self) -> AttackTree: """Build the attack tree.""" if not self._root: raise ValueError("No root goal defined") return AttackTree( name=self.name, description=self.description, root=self._root ) def _current(self) -> AttackNode: if not self._node_stack: raise ValueError("No current node") return self._node_stack[-1] # Example usage def build_account_takeover_tree() -> AttackTree: """Build attack tree for account takeover scenario.""" return ( AttackTreeBuilder("Account Takeover", "Gain unauthorized access to user account") .goal("G1", "Take Over User Account") .or_node("S1", "Steal Credentials") .attack( "A1", "Phishing Attack", difficulty=Difficulty.LOW, cost=Cost.LOW, detection=DetectionRisk.MEDIUM, mitigations=["Security awareness training", "Email filtering"] ) .attack( "A2", "Credential Stuffing", difficulty=Difficulty.TRIVIAL, cost=Cost.LOW, detection=DetectionRisk.HIGH, mitigations=["Rate limiting", "MFA", "Password breach monitoring"] ) .attack( "A3", "Keylogger Malware", difficulty=Difficulty.MEDIUM, cost=Cost.MEDIUM, detection=DetectionRisk.MEDIUM, mitigations=["Endpoint protection", "MFA"] ) .end() .or_node("S2", "Bypass Authentication") .attack( "A4", "Session Hijacking", difficulty=Difficulty.MEDIUM, cost=Cost.LOW, detection=DetectionRisk.LOW, mitigations=["Secure session management", "HTTPS only"] ) .attack( "A5", "Authentication Bypass Vulnerability", difficulty=Difficulty.HIGH, cost=Cost.LOW, detection=DetectionRisk.LOW, mitigations=["Security testing", "Code review", "WAF"] ) .end() .or_node("S3", "Social Engineering") .and_node("S3.1", "Account Recovery Attack") .attack( "A6", "Gather Personal Information", difficulty=Difficulty.LOW, cost=Cost.FREE, detection=DetectionRisk.NONE ) .attack( "A7", "Call Support Desk", difficulty=Difficulty.MEDIUM, cost=Cost.FREE, detection=DetectionRisk.MEDIUM, mitigations=["Support verification procedures", "Security questions"] ) .end() .end() .build() ) ``` ### Template 3: Mermaid Diagram Generator ```python class MermaidExporter: """Export attack trees to Mermaid diagram format.""" def __init__(self, tree: AttackTree): self.tree = tree self._lines: List[str] = [] self._node_count = 0 def export(self) -> str: """Export tree to Mermaid flowchart.""" self._lines = ["flowchart TD"] self._export_node(self.tree.root, None) return "\n".join(self._lines) def _export_node(self, node: AttackNode, parent_id: Optional[str]) -> str: """Recursively export nodes.""" node_id = f"N{self._node_count}" self._node_count += 1 # Node shape based on type if node.node_type == NodeType.OR: shape = f"{node_id}(({node.name}))" elif node.node_type == NodeType.AND: shape = f"{node_id}[{node.name}]" else: # LEAF # Color based on difficulty style = self._get_leaf_style(node) shape = f"{node_id}[/{node.name}/]" self._lines.append(f" style {node_id} {style}") self._lines.append(f" {shape}") if parent_id: connector = "-->" if node.node_type != NodeType.AND else "==>" self._lines.append(f" {parent_id} {connector} {node_id}") for child in node.children: self._export_node(child, node_id) return node_id def _get_leaf_style(self, node: AttackNode) -> str: """Get style based on attack attributes.""" colors = { Difficulty.TRIVIAL: "fill:#ff6b6b", # Red - easy attack Difficulty.LOW: "fill:#ffa06b", Difficulty.MEDIUM: "fill:#ffd93d", Difficulty.HIGH: "fill:#6bcb77", Difficulty.EXPERT: "fill:#4d96ff", # Blue - hard attack } color = colors.get(node.attributes.difficulty, "fill:#gray") return color class PlantUMLExporter: """Export attack trees to PlantUML format.""" def __init__(self, tree: AttackTree): self.tree = tree def export(self) -> str: """Export tree to PlantUML.""" lines = [ "@startmindmap", f"* {self.tree.name}", ] self._export_node(self.tree.root, lines, 1) lines.append("@endmindmap") return "\n".join(lines) def _export_node(self, node: AttackNode, lines: List[str], depth: int) -> None: """Recursively export nodes.""" prefix = "*" * (depth + 1) if node.node_type == NodeType.OR: marker = "[OR]" elif node.node_type == NodeType.AND: marker = "[AND]" else: diff = node.attributes.difficulty.name marker = f"<<{diff}>>" lines.append(f"{prefix} {marker} {node.name}") for child in node.children: self._export_node(child, lines, depth + 1) ``` ### Template 4: Attack Path Analysis ```python from typing import Set, Tuple class AttackPathAnalyzer: """Analyze attack paths and coverage.""" def __init__(self, tree: AttackTree): self.tree = tree def get_all_paths(self) -> List[List[AttackNode]]: """Get all possible attack paths.""" paths = [] self._collect_paths(self.tree.root, [], paths) return paths def _collect_paths( self, node: AttackNode, current_path: List[AttackNode], all_paths: List[List[AttackNode]] ) -> None: """Recursively collect all paths.""" current_path = current_path + [node] if node.node_type == NodeType.LEAF: all_paths.append(current_path) return if not node.children: all_paths.append(current_path) return if node.node_type == NodeType.OR: # Each child is a separate path for child in node.children: self._collect_paths(child, current_path, all_paths) else: # AND # Must combine all children child_paths = [] for child in node.children: child_sub_paths = [] self._collect_paths(child, [], child_sub_paths) child_paths.append(child_sub_paths) # Combine paths from all AND children combined = self._combine_and_paths(child_paths) for combo in combined: all_paths.append(current_path + combo) def _combine_and_paths( self, child_paths: List[List[List[AttackNode]]] ) -> List[List[AttackNode]]: """Combine paths from AND node children.""" if not child_paths: return [[]] if len(child_paths) == 1: return [path for paths in child_paths for path in paths] # Cartesian product of all child path combinations result = [[]] for paths in child_paths: new_result = [] for existing in result: for path in paths: new_result.append(existing + path) result = new_result return result def calculate_path_metrics(self, path: List[AttackNode]) -> Dict: """Calculate metrics for a specific path.""" leaves = [n for n in path if n.node_type == NodeType.LEAF] total_difficulty = sum(n.attributes.difficulty.value for n in leaves) total_cost = sum(n.attributes.cost.value for n in leaves) total_time = sum(n.attributes.time_hours for n in leaves) max_detection = max((n.attributes.detection_risk.value for n in leaves), default=0) return { "steps": len(leaves), "total_difficulty": total_difficulty, "avg_difficulty": total_difficulty / len(leaves) if leaves else 0, "total_cost": total_cost, "total_time_hours": total_time, "max_detection_risk": max_detection, "requires_insider": any(n.attributes.requires_insider for n in leaves), "requires_physical": any(n.attributes.requires_physical for n in leaves), } def identify_critical_nodes(self) -> List[Tuple[AttackNode, int]]: """Find nodes that appear in the most paths.""" paths = self.get_all_paths() node_counts: Dict[str, Tuple[AttackNode, int]] = {} for path in paths: for node in path: if node.id not in node_counts: node_counts[node.id] = (node, 0) node_counts[node.id] = (node, node_counts[node.id][1] + 1) return sort
πŸ‘0
πŸ‘οΈ0
πŸ€– Auto-discovered
πŸ€–system promptβ€’7 months ago

security-requirement-extraction

Derive security requirements from threat models and business

security
⭐1
# Security Requirement Extraction Transform threat analysis into actionable security requirements. ## When to Use This Skill - Converting threat models to requirements - Writing security user stories - Creating security test cases - Building security acceptance criteria - Compliance requirement mapping - Security architecture documentation ## Core Concepts ### 1. Requirement Categories ``` Business Requirements β†’ Security Requirements β†’ Technical Controls ↓ ↓ ↓ "Protect customer "Encrypt PII at rest" "AES-256 encryption data" with KMS key rotation" ``` ### 2. Security Requirement Types | Type | Focus | Example | | ------------------ | ----------------------- | ------------------------------------- | | **Functional** | What system must do | "System must authenticate users" | | **Non-functional** | How system must perform | "Authentication must complete in <2s" | | **Constraint** | Limitations imposed | "Must use approved crypto libraries" | ### 3. Requirement Attributes | Attribute | Description | | ---------------- | --------------------------- | | **Traceability** | Links to threats/compliance | | **Testability** | Can be verified | | **Priority** | Business importance | | **Risk Level** | Impact if not met | ## Templates ### Template 1: Security Requirement Model ```python from dataclasses import dataclass, field from enum import Enum from typing import List, Dict, Optional, Set from datetime import datetime class RequirementType(Enum): FUNCTIONAL = "functional" NON_FUNCTIONAL = "non_functional" CONSTRAINT = "constraint" class Priority(Enum): CRITICAL = 1 HIGH = 2 MEDIUM = 3 LOW = 4 class SecurityDomain(Enum): AUTHENTICATION = "authentication" AUTHORIZATION = "authorization" DATA_PROTECTION = "data_protection" AUDIT_LOGGING = "audit_logging" INPUT_VALIDATION = "input_validation" ERROR_HANDLING = "error_handling" SESSION_MANAGEMENT = "session_management" CRYPTOGRAPHY = "cryptography" NETWORK_SECURITY = "network_security" AVAILABILITY = "availability" class ComplianceFramework(Enum): PCI_DSS = "pci_dss" HIPAA = "hipaa" GDPR = "gdpr" SOC2 = "soc2" NIST_CSF = "nist_csf" ISO_27001 = "iso_27001" OWASP = "owasp" @dataclass class SecurityRequirement: id: str title: str description: str req_type: RequirementType domain: SecurityDomain priority: Priority rationale: str = "" acceptance_criteria: List[str] = field(default_factory=list) test_cases: List[str] = field(default_factory=list) threat_refs: List[str] = field(default_factory=list) compliance_refs: List[str] = field(default_factory=list) dependencies: List[str] = field(default_factory=list) status: str = "draft" owner: str = "" created_date: datetime = field(default_factory=datetime.now) def to_user_story(self) -> str: """Convert to user story format.""" return f""" **{self.id}: {self.title}** As a security-conscious system, I need to {self.description.lower()}, So that {self.rationale.lower()}. **Acceptance Criteria:** {chr(10).join(f'- [ ] {ac}' for ac in self.acceptance_criteria)} **Priority:** {self.priority.name} **Domain:** {self.domain.value} **Threat References:** {', '.join(self.threat_refs)} """ def to_test_spec(self) -> str: """Convert to test specification.""" return f""" ## Test Specification: {self.id} ### Requirement {self.description} ### Test Cases {chr(10).join(f'{i+1}. {tc}' for i, tc in enumerate(self.test_cases))} ### Acceptance Criteria Verification {chr(10).join(f'- {ac}' for ac in self.acceptance_criteria)} """ @dataclass class RequirementSet: name: str version: str requirements: List[SecurityRequirement] = field(default_factory=list) def add(self, req: SecurityRequirement) -> None: self.requirements.append(req) def get_by_domain(self, domain: SecurityDomain) -> List[SecurityRequirement]: return [r for r in self.requirements if r.domain == domain] def get_by_priority(self, priority: Priority) -> List[SecurityRequirement]: return [r for r in self.requirements if r.priority == priority] def get_by_threat(self, threat_id: str) -> List[SecurityRequirement]: return [r for r in self.requirements if threat_id in r.threat_refs] def get_critical_requirements(self) -> List[SecurityRequirement]: return [r for r in self.requirements if r.priority == Priority.CRITICAL] def export_markdown(self) -> str: """Export all requirements as markdown.""" lines = [f"# Security Requirements: {self.name}\n"] lines.append(f"Version: {self.version}\n") for domain in SecurityDomain: domain_reqs = self.get_by_domain(domain) if domain_reqs: lines.append(f"\n## {domain.value.replace('_', ' ').title()}\n") for req in domain_reqs: lines.append(req.to_user_story()) return "\n".join(lines) def traceability_matrix(self) -> Dict[str, List[str]]: """Generate threat-to-requirement traceability.""" matrix = {} for req in self.requirements: for threat_id in req.threat_refs: if threat_id not in matrix: matrix[threat_id] = [] matrix[threat_id].append(req.id) return matrix ``` ### Template 2: Threat-to-Requirement Extractor ```python from dataclasses import dataclass from typing import List, Dict, Tuple @dataclass class ThreatInput: id: str category: str # STRIDE category title: str description: str target: str impact: str likelihood: str class RequirementExtractor: """Extract security requirements from threats.""" # Mapping of STRIDE categories to security domains and requirement patterns STRIDE_MAPPINGS = { "SPOOFING": { "domains": [SecurityDomain.AUTHENTICATION, SecurityDomain.SESSION_MANAGEMENT], "patterns": [ ("Implement strong authentication for {target}", "Ensure {target} authenticates all users before granting access"), ("Validate identity tokens for {target}", "All authentication tokens must be cryptographically verified"), ("Implement session management for {target}", "Sessions must be securely managed with proper expiration"), ] }, "TAMPERING": { "domains": [SecurityDomain.INPUT_VALIDATION, SecurityDomain.DATA_PROTECTION], "patterns": [ ("Validate all input to {target}", "All input must be validated against expected formats"), ("Implement integrity checks for {target}", "Data integrity must be verified using cryptographic signatures"), ("Protect {target} from modification", "Implement controls to prevent unauthorized data modification"), ] }, "REPUDIATION": { "domains": [SecurityDomain.AUDIT_LOGGING], "patterns": [ ("Log all security events for {target}", "Security-relevant events must be logged for audit purposes"), ("Implement non-repudiation for {target}", "Critical actions must have cryptographic proof of origin"), ("Protect audit logs for {target}", "Audit logs must be tamper-evident and protected"), ] }, "INFORMATION_DISCLOSURE": { "domains": [SecurityDomain.DATA_PROTECTION, SecurityDomain.CRYPTOGRAPHY], "patterns": [ ("Encrypt sensitive data in {target}", "Sensitive data must be encrypted at rest and in transit"), ("Implement access controls for {target}", "Data access must be restricted based on need-to-know"), ("Prevent information leakage from {target}", "Error messages and logs must not expose sensitive information"), ] }, "DENIAL_OF_SERVICE": { "domains": [SecurityDomain.AVAILABILITY, SecurityDomain.INPUT_VALIDATION], "patterns": [ ("Implement rate limiting for {target}", "Requests must be rate-limited to prevent resource exhaustion"), ("Ensure availability of {target}", "System must remain available under high load conditions"), ("Implement resource quotas for {target}", "Resource consumption must be bounded and monitored"), ] }, "ELEVATION_OF_PRIVILEGE": { "domains": [SecurityDomain.AUTHORIZATION], "patterns": [ ("Enforce authorization for {target}", "All actions must be authorized based on user permissions"), ("Implement least privilege for {target}", "Users must only have minimum necessary permissions"), ("Validate permissions for {target}", "Permission checks must be performed server-side"), ] }, } def extract_requirements( self, threats: List[ThreatInput], project_name: str ) -> RequirementSet: """Extract security requirements from threats.""" req_set = RequirementSet( name=f"{project_name} Security Requirements", version="1.0" ) req_counter = 1 for threat in threats: reqs = self._threat_to_requirements(threat, req_counter) for req in reqs: req_set.add(req) req_counter += len(reqs) return req_set def _threat_to_requirements( self, threat: ThreatInput, start_id: int ) -> List[SecurityRequirement]: """Convert a single threat to requirements.""" requirements = [] mapping = self.STRIDE_MAPPINGS.get(threat.category, {}) domains = mapping.get("domains", []) patterns = mapping.get("patterns", []) priority = self._calculate_priority(threat.impact, threat.likelihood) for i, (title_pattern, desc_pattern) in enumerate(patterns): req = SecurityRequirement( id=f"SR-{start_id + i:03d}", title=title_pattern.format(target=threat.target), description=desc_pattern.format(target=threat.target), req_type=RequirementType.FUNCTIONAL, domain=domains[i % len(domains)] if domains else SecurityDomain.DATA_PROTECTION, priority=priority, rationale=f"Mitigates threat: {threat.title}", threat_refs=[threat.id], acceptance_criteria=self._generate_acceptance_criteria( threat.category, threat.target ), test_cases=self._generate_test_cases( threat.category, threat.target ) ) requirements.append(req) return requirements def _calculate_priority(self, impact: str, likelihood: str) -> Priority: """Calculate requirement priority from threat attributes.""" score_map = {"LOW": 1, "MEDIUM": 2, "HIGH": 3, "CRITICAL": 4} impact_score = score_map.get(impact.upper(), 2) likelihood_score = score_map.get(likelihood.upper(), 2) combined = impact_score * likelihood_score if combined >= 12: return Priority.CRITICAL elif combined >= 6: return Priority.HIGH elif combined >= 3: return Priority.MEDIUM return Priority.LOW def _generate_acceptance_criteria( self, category: str, target: str ) -> List[str]: """Generate acceptance criteria for requirement.""" criteria_templates = { "SPOOFING": [ f"Users must authenticate before accessing {target}", "Authentication failures are logged and monitored", "Multi-factor authentication is available for sensitive operations", ], "TAMPERING": [ f"All input to {target} is validated", "Data integrity is verified before processing", "Modification attempts trigger alerts", ], "REPUDIATION": [ f"All actions on {target} are logged with user identity", "Logs cannot be modified by regular users", "Log retention meets compliance requirements", ], "INFORMATION_DISCLOSURE": [ f"Sensitive data in {target} is encrypted", "Access to sensitive data is logged", "Error messages do not reveal sensitive information", ], "DENIAL_OF_SERVICE": [ f"Rate limiting is enforced on {target}", "System degrades gracefully under load", "Resource exhaustion triggers alerts", ], "ELEVATION_OF_PRIVILEGE": [ f"Authorization is checked for all {target} operations", "Users cannot access resources beyond their permissions", "Privilege changes are logged and monitored", ], } return criteria_templates.get(category, []) def _generate_test_cases( self, category: str, target: str ) -> List[str]: """Generate test cases for requirement.""" test_templates = { "SPOOFING": [ f"Test: Unauthenticated access to {target} is denied", "Test: Invalid credentials are rejected", "Test: Session tokens cannot be forged", ], "TAMPERING": [ f"Test: Invalid input to {target} is rejected", "Test: Tampered data is detected and rejected", "Test: SQL injection attempts are blocked", ], "REPUDIATION": [ "Test: Security events are logged", "Test: Logs include sufficient detail for forensics", "Test: Log integrity is protected", ], "INFORMATION_DISCLOSURE": [ f"Test: {target} data is encrypted in transit", f"Test: {target} data is encrypted at rest", "Test: Error messages are sanitized", ], "DENIAL_OF_SERVICE": [ f"Test: Rate limiting on {target} works correctly", "Test: System handles burst traffic gracefully", "Test: Resource limits are enforced", ], "ELEVATION_OF_PRIVILEGE": [ f"Test: Unauthorized access to {target} is denied", "Test: Privilege escalation attempts are blocked", "Test: IDOR vulnerabilities are not present", ], } return test_templates.get(category, []) ``` ### Template 3: Compliance Mapping ```python from typing import Dict, List, Set class ComplianceMapper: """Map security requirements to compliance frameworks.""" FRAMEWORK_CONTROLS = { ComplianceFramework.PCI_DSS: { SecurityDomain.AUTHENTICATION: ["8.1", "8.2", "8.3"], SecurityDomain.AUTHORIZATION: ["7.1", "7.2"], SecurityDomain.DATA_PROTECTION: ["3.4", "3.5", "4.1"], SecurityDomain.AUDIT_LOGGING: ["10.1", "10.2", "10.3"], SecurityDomain.NETWORK_SECURITY: ["1.1", "1.2", "1.3"], SecurityDomain.CRYPTOGRAPHY: ["3.5", "3.6", "4.1"], }, ComplianceFramework.HIPAA: { SecurityDomain.AUTHENTICATION: ["164.312(d)"], SecurityDomain.AUTHORIZATION: ["164.312(a)(1)"], SecurityDomain.DATA_PROTECTION: ["164.312(a)(2)(iv)", "164.312(e)(2)(ii)"], SecurityDomain.AUDIT_LOGGING: ["164.312(b)"], }, ComplianceFramework.GDPR: { SecurityDomain.DATA_PROTECTION: ["Art. 32", "Art. 25"], SecurityDomain.AUDIT_LOGGING: ["Art. 30"], SecurityDomain.AUTHORIZATION: ["Art. 25"], }, ComplianceFramework.OWASP: { SecurityDomain.AUTHENTICATION: ["V2.1", "V2.2", "V2.3"], SecurityDomain.SESSION_MANAGEMENT: ["V3.1", "V3.2", "V3.3"], SecurityDomain.INPUT_VALIDATION: ["V5.1", "V5.2", "V5.3"], SecurityDomain.CRYPTOGRAPHY: ["V6.1", "V6.2"], SecurityDomain.ERROR_HANDLING: ["V7.1", "V7.2"], SecurityDomain.DATA_PROTECTION: ["V8.1", "V8.2", "V8.3"], SecurityDomain.AUDIT_LOGGING: ["V7.1", "V7.2"], }, } def map_requirement_to_compliance( self, requirement: SecurityRequirement, frameworks: List[ComplianceFramework] ) -> Dict[str, List[str]]: """Map a requirement to compliance controls.""" mapping = {} for framework in frameworks: controls = self.FRAMEWORK_CONTROLS.get(framework, {}) domain_controls = controls.get(requirement.domain, []) if domain_controls: mapping[framework.value] = domain_controls return mapping def get_requirements_for_control( self, requirement_set: RequirementSet, framework: ComplianceFramework, control_id: str ) -> List[SecurityRequirement]: """Find requirements that satisfy a compliance control.""" matching = [] framework_controls = self.FRAMEWORK_CONTROLS.get(framework, {}) for domain, controls in framework_controls.items(): if control_id in controls: matching.extend(requirement_set.get_by_domain(domain)) return matching def generate_compliance_matrix( self, requirement_set: RequirementSet, frameworks: List[ComplianceFramework] ) -> Dict[str, Dict[str, List[str]]]: """Generate compliance traceability matrix.""" matrix = {} for framework in frameworks: matrix[framework.value] = {} framework_controls = self.FRAMEWORK_CONTROLS.get(framework, {}) for domain, controls in framework_controls.items(): for control in controls: reqs = self.get_requirements_for_control( requirement_set, framework, control ) if reqs: matrix[framework.value][control] = [r.id for r in reqs] return matrix def gap_analysis( self, requirement_set: RequirementSet, framework: ComplianceFramework ) -> Dict[str, List[str]]: """Identify compliance gaps.""" gaps = {"missing_controls": [], "weak_coverage": []} framework_controls = self.FRAMEWORK_CONTROLS.get(framework, {}) for domain, controls in framework_controls.items(): domain_reqs = requirement_set.get_by_domain(domain) for
πŸ‘0
πŸ‘οΈ0
πŸ€– Auto-discovered
πŸ€–system promptβ€’7 months ago

stride-analysis-patterns

Apply STRIDE methodology to systematically identify threats. Use

security
⭐1
# STRIDE Analysis Patterns Systematic threat identification using the STRIDE methodology. ## When to Use This Skill - Starting new threat modeling sessions - Analyzing existing system architecture - Reviewing security design decisions - Creating threat documentation - Training teams on threat identification - Compliance and audit preparation ## Core Concepts ### 1. STRIDE Categories ``` S - Spoofing β†’ Authentication threats T - Tampering β†’ Integrity threats R - Repudiation β†’ Non-repudiation threats I - Information β†’ Confidentiality threats Disclosure D - Denial of β†’ Availability threats Service E - Elevation of β†’ Authorization threats Privilege ``` ### 2. Threat Analysis Matrix | Category | Question | Control Family | | ------------------- | ----------------------------------------- | -------------- | | **Spoofing** | Can attacker pretend to be someone else? | Authentication | | **Tampering** | Can attacker modify data in transit/rest? | Integrity | | **Repudiation** | Can attacker deny actions? | Logging/Audit | | **Info Disclosure** | Can attacker access unauthorized data? | Encryption | | **DoS** | Can attacker disrupt availability? | Rate limiting | | **Elevation** | Can attacker gain higher privileges? | Authorization | ## Templates ### Template 1: STRIDE Threat Model Document ```markdown # Threat Model: [System Name] ## 1. System Overview ### 1.1 Description [Brief description of the system and its purpose] ### 1.2 Data Flow Diagram ``` [User] --> [Web App] --> [API Gateway] --> [Backend Services] | v [Database] ``` ### 1.3 Trust Boundaries - **External Boundary**: Internet to DMZ - **Internal Boundary**: DMZ to Internal Network - **Data Boundary**: Application to Database ## 2. Assets | Asset | Sensitivity | Description | |-------|-------------|-------------| | User Credentials | High | Authentication tokens, passwords | | Personal Data | High | PII, financial information | | Session Data | Medium | Active user sessions | | Application Logs | Medium | System activity records | | Configuration | High | System settings, secrets | ## 3. STRIDE Analysis ### 3.1 Spoofing Threats | ID | Threat | Target | Impact | Likelihood | |----|--------|--------|--------|------------| | S1 | Session hijacking | User sessions | High | Medium | | S2 | Token forgery | JWT tokens | High | Low | | S3 | Credential stuffing | Login endpoint | High | High | **Mitigations:** - [ ] Implement MFA - [ ] Use secure session management - [ ] Implement account lockout policies ### 3.2 Tampering Threats | ID | Threat | Target | Impact | Likelihood | |----|--------|--------|--------|------------| | T1 | SQL injection | Database queries | Critical | Medium | | T2 | Parameter manipulation | API requests | High | High | | T3 | File upload abuse | File storage | High | Medium | **Mitigations:** - [ ] Input validation on all endpoints - [ ] Parameterized queries - [ ] File type validation ### 3.3 Repudiation Threats | ID | Threat | Target | Impact | Likelihood | |----|--------|--------|--------|------------| | R1 | Transaction denial | Financial ops | High | Medium | | R2 | Access log tampering | Audit logs | Medium | Low | | R3 | Action attribution | User actions | Medium | Medium | **Mitigations:** - [ ] Comprehensive audit logging - [ ] Log integrity protection - [ ] Digital signatures for critical actions ### 3.4 Information Disclosure Threats | ID | Threat | Target | Impact | Likelihood | |----|--------|--------|--------|------------| | I1 | Data breach | User PII | Critical | Medium | | I2 | Error message leakage | System info | Low | High | | I3 | Insecure transmission | Network traffic | High | Medium | **Mitigations:** - [ ] Encryption at rest and in transit - [ ] Sanitize error messages - [ ] Implement TLS 1.3 ### 3.5 Denial of Service Threats | ID | Threat | Target | Impact | Likelihood | |----|--------|--------|--------|------------| | D1 | Resource exhaustion | API servers | High | High | | D2 | Database overload | Database | Critical | Medium | | D3 | Bandwidth saturation | Network | High | Medium | **Mitigations:** - [ ] Rate limiting - [ ] Auto-scaling - [ ] DDoS protection ### 3.6 Elevation of Privilege Threats | ID | Threat | Target | Impact | Likelihood | |----|--------|--------|--------|------------| | E1 | IDOR vulnerabilities | User resources | High | High | | E2 | Role manipulation | Admin access | Critical | Low | | E3 | JWT claim tampering | Authorization | High | Medium | **Mitigations:** - [ ] Proper authorization checks - [ ] Principle of least privilege - [ ] Server-side role validation ## 4. Risk Assessment ### 4.1 Risk Matrix ``` IMPACT Low Med High Crit Low 1 2 3 4 L Med 2 4 6 8 I High 3 6 9 12 K Crit 4 8 12 16 ``` ### 4.2 Prioritized Risks | Rank | Threat | Risk Score | Priority | |------|--------|------------|----------| | 1 | SQL Injection (T1) | 12 | Critical | | 2 | IDOR (E1) | 9 | High | | 3 | Credential Stuffing (S3) | 9 | High | | 4 | Data Breach (I1) | 8 | High | ## 5. Recommendations ### Immediate Actions 1. Implement input validation framework 2. Add rate limiting to authentication endpoints 3. Enable comprehensive audit logging ### Short-term (30 days) 1. Deploy WAF with OWASP ruleset 2. Implement MFA for sensitive operations 3. Encrypt all PII at rest ### Long-term (90 days) 1. Security awareness training 2. Penetration testing 3. Bug bounty program ``` ### Template 2: STRIDE Analysis Code ```python from dataclasses import dataclass, field from enum import Enum from typing import List, Dict, Optional import json class StrideCategory(Enum): SPOOFING = "S" TAMPERING = "T" REPUDIATION = "R" INFORMATION_DISCLOSURE = "I" DENIAL_OF_SERVICE = "D" ELEVATION_OF_PRIVILEGE = "E" class Impact(Enum): LOW = 1 MEDIUM = 2 HIGH = 3 CRITICAL = 4 class Likelihood(Enum): LOW = 1 MEDIUM = 2 HIGH = 3 CRITICAL = 4 @dataclass class Threat: id: str category: StrideCategory title: str description: str target: str impact: Impact likelihood: Likelihood mitigations: List[str] = field(default_factory=list) status: str = "open" @property def risk_score(self) -> int: return self.impact.value * self.likelihood.value @property def risk_level(self) -> str: score = self.risk_score if score >= 12: return "Critical" elif score >= 6: return "High" elif score >= 3: return "Medium" return "Low" @dataclass class Asset: name: str sensitivity: str description: str data_classification: str @dataclass class TrustBoundary: name: str description: str from_zone: str to_zone: str @dataclass class ThreatModel: name: str version: str description: str assets: List[Asset] = field(default_factory=list) boundaries: List[TrustBoundary] = field(default_factory=list) threats: List[Threat] = field(default_factory=list) def add_threat(self, threat: Threat) -> None: self.threats.append(threat) def get_threats_by_category(self, category: StrideCategory) -> List[Threat]: return [t for t in self.threats if t.category == category] def get_critical_threats(self) -> List[Threat]: return [t for t in self.threats if t.risk_level in ("Critical", "High")] def generate_report(self) -> Dict: """Generate threat model report.""" return { "summary": { "name": self.name, "version": self.version, "total_threats": len(self.threats), "critical_threats": len([t for t in self.threats if t.risk_level == "Critical"]), "high_threats": len([t for t in self.threats if t.risk_level == "High"]), }, "by_category": { cat.name: len(self.get_threats_by_category(cat)) for cat in StrideCategory }, "top_risks": [ { "id": t.id, "title": t.title, "risk_score": t.risk_score, "risk_level": t.risk_level } for t in sorted(self.threats, key=lambda x: x.risk_score, reverse=True)[:10] ] } class StrideAnalyzer: """Automated STRIDE analysis helper.""" STRIDE_QUESTIONS = { StrideCategory.SPOOFING: [ "Can an attacker impersonate a legitimate user?", "Are authentication tokens properly validated?", "Can session identifiers be predicted or stolen?", "Is multi-factor authentication available?", ], StrideCategory.TAMPERING: [ "Can data be modified in transit?", "Can data be modified at rest?", "Are input validation controls sufficient?", "Can an attacker manipulate application logic?", ], StrideCategory.REPUDIATION: [ "Are all security-relevant actions logged?", "Can logs be tampered with?", "Is there sufficient attribution for actions?", "Are timestamps reliable and synchronized?", ], StrideCategory.INFORMATION_DISCLOSURE: [ "Is sensitive data encrypted at rest?", "Is sensitive data encrypted in transit?", "Can error messages reveal sensitive information?", "Are access controls properly enforced?", ], StrideCategory.DENIAL_OF_SERVICE: [ "Are rate limits implemented?", "Can resources be exhausted by malicious input?", "Is there protection against amplification attacks?", "Are there single points of failure?", ], StrideCategory.ELEVATION_OF_PRIVILEGE: [ "Are authorization checks performed consistently?", "Can users access other users' resources?", "Can privilege escalation occur through parameter manipulation?", "Is the principle of least privilege followed?", ], } def generate_questionnaire(self, component: str) -> List[Dict]: """Generate STRIDE questionnaire for a component.""" questionnaire = [] for category, questions in self.STRIDE_QUESTIONS.items(): for q in questions: questionnaire.append({ "component": component, "category": category.name, "question": q, "answer": None, "notes": "" }) return questionnaire def suggest_mitigations(self, category: StrideCategory) -> List[str]: """Suggest common mitigations for a STRIDE category.""" mitigations = { StrideCategory.SPOOFING: [ "Implement multi-factor authentication", "Use secure session management", "Implement account lockout policies", "Use cryptographically secure tokens", "Validate authentication at every request", ], StrideCategory.TAMPERING: [ "Implement input validation", "Use parameterized queries", "Apply integrity checks (HMAC, signatures)", "Implement Content Security Policy", "Use immutable infrastructure", ], StrideCategory.REPUDIATION: [ "Enable comprehensive audit logging", "Protect log integrity", "Implement digital signatures", "Use centralized, tamper-evident logging", "Maintain accurate timestamps", ], StrideCategory.INFORMATION_DISCLOSURE: [ "Encrypt data at rest and in transit", "Implement proper access controls", "Sanitize error messages", "Use secure defaults", "Implement data classification", ], StrideCategory.DENIAL_OF_SERVICE: [ "Implement rate limiting", "Use auto-scaling", "Deploy DDoS protection", "Implement circuit breakers", "Set resource quotas", ], StrideCategory.ELEVATION_OF_PRIVILEGE: [ "Implement proper authorization", "Follow principle of least privilege", "Validate permissions server-side", "Use role-based access control", "Implement security boundaries", ], } return mitigations.get(category, []) ``` ### Template 3: Data Flow Diagram Analysis ```python from dataclasses import dataclass from typing import List, Set, Tuple from enum import Enum class ElementType(Enum): EXTERNAL_ENTITY = "external" PROCESS = "process" DATA_STORE = "datastore" DATA_FLOW = "dataflow" @dataclass class DFDElement: id: str name: str type: ElementType trust_level: int # 0 = untrusted, higher = more trusted description: str = "" @dataclass class DataFlow: id: str name: str source: str destination: str data_type: str protocol: str encrypted: bool = False class DFDAnalyzer: """Analyze Data Flow Diagrams for STRIDE threats.""" def __init__(self): self.elements: Dict[str, DFDElement] = {} self.flows: List[DataFlow] = [] def add_element(self, element: DFDElement) -> None: self.elements[element.id] = element def add_flow(self, flow: DataFlow) -> None: self.flows.append(flow) def find_trust_boundary_crossings(self) -> List[Tuple[DataFlow, int]]: """Find data flows that cross trust boundaries.""" crossings = [] for flow in self.flows: source = self.elements.get(flow.source) dest = self.elements.get(flow.destination) if source and dest and source.trust_level != dest.trust_level: trust_diff = abs(source.trust_level - dest.trust_level) crossings.append((flow, trust_diff)) return sorted(crossings, key=lambda x: x[1], reverse=True) def identify_threats_per_element(self) -> Dict[str, List[StrideCategory]]: """Map applicable STRIDE categories to element types.""" threat_mapping = { ElementType.EXTERNAL_ENTITY: [ StrideCategory.SPOOFING, StrideCategory.REPUDIATION, ], ElementType.PROCESS: [ StrideCategory.SPOOFING, StrideCategory.TAMPERING, StrideCategory.REPUDIATION, StrideCategory.INFORMATION_DISCLOSURE, StrideCategory.DENIAL_OF_SERVICE, StrideCategory.ELEVATION_OF_PRIVILEGE, ], ElementType.DATA_STORE: [ StrideCategory.TAMPERING, StrideCategory.REPUDIATION, StrideCategory.INFORMATION_DISCLOSURE, StrideCategory.DENIAL_OF_SERVICE, ], ElementType.DATA_FLOW: [ StrideCategory.TAMPERING, StrideCategory.INFORMATION_DISCLOSURE, StrideCategory.DENIAL_OF_SERVICE, ], } result = {} for elem_id, elem in self.elements.items(): result[elem_id] = threat_mapping.get(elem.type, []) return result def analyze_unencrypted_flows(self) -> List[DataFlow]: """Find unencrypted data flows crossing trust boundaries.""" risky_flows = [] for flow in self.flows: if not flow.encrypted: source = self.elements.get(flow.source) dest = self.elements.get(flow.destination) if source and dest and source.trust_level != dest.trust_level: risky_flows.append(flow) return risky_flows def generate_threat_enumeration(self) -> List[Dict]: """Generate comprehensive threat enumeration.""" threats = [] element_threats = self.identify_threats_per_element() for elem_id, categories in element_threats.items(): elem = self.elements[elem_id] for category in categories: threats.append({ "element_id": elem_id, "element_name": elem.name, "element_type": elem.type.value, "stride_category": category.name, "description": f"{category.name} threat against {elem.name}", "trust_level": elem.trust_level }) return threats ``` ### Template 4: STRIDE per Interaction ```python from typing import List, Dict, Optional from dataclasses import dataclass @dataclass class Interaction: """Represents an interaction between two components.""" id: str source: str target: str action: str data: str protocol: str class StridePerInteraction: """Apply STRIDE to each interaction in the system.""" INTERACTION_THREATS = { # Source type -> Target type -> Applicable threats ("external", "process"): { "S": "External entity spoofing identity to process", "T": "Tampering with data sent to process", "R": "External entity denying sending data", "I": "Data exposure during transmission", "D": "Flooding process with requests", "E": "Exploiting process to gain privileges", }, ("process", "datastore"): { "T": "Process tampering with stored data", "R": "Process denying data modifications", "I": "Unauthorized data access by process", "D": "Process exhausting storage resources", }, ("process", "process"): { "S": "Process spoofing another process", "T": "Tampering with inter-process data", "I": "Data leakage between processes", "D": "One process overwhelming another", "E": "Process gaining elevated access", }, } def analyze_interaction( self, interaction: Interaction, source_type: str, target_type: str ) -> List[Dict]: """Analyze a single interaction for STRIDE threats.""" threats = [] key = (source_type, target_type) applicable_threats = self.INTERACTION_THREATS.get(key, {}) for stride_code, description in applicable_threats.items(): threats.append({ "interaction_id": interaction.id, "source": interaction.source, "target": interaction.target, "stride_category": stride_code, "threat_description": description, "context": f"{interaction.action} - {interaction.data}", }) return threats def generate_threat_matrix( self, interactions: List[Interaction], element_types: Dict[str, str] ) -> List[Dict]:
πŸ‘0
πŸ‘οΈ0
πŸ€– Auto-discovered
πŸ€–system promptβ€’7 months ago

threat-mitigation-mapping

Map identified threats to appropriate security controls and

security
⭐1
# Threat Mitigation Mapping Connect threats to controls for effective security planning. ## When to Use This Skill - Prioritizing security investments - Creating remediation roadmaps - Validating control coverage - Designing defense-in-depth - Security architecture review - Risk treatment planning ## Core Concepts ### 1. Control Categories ``` Preventive ────► Stop attacks before they occur β”‚ (Firewall, Input validation) β”‚ Detective ─────► Identify attacks in progress β”‚ (IDS, Log monitoring) β”‚ Corrective ────► Respond and recover from attacks (Incident response, Backup restore) ``` ### 2. Control Layers | Layer | Examples | | --------------- | ------------------------------------ | | **Network** | Firewall, WAF, DDoS protection | | **Application** | Input validation, authentication | | **Data** | Encryption, access controls | | **Endpoint** | EDR, patch management | | **Process** | Security training, incident response | ### 3. Defense in Depth ``` β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” β”‚ Perimeter β”‚ ← Firewall, WAF β”‚ β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” β”‚ β”‚ β”‚ Network β”‚ β”‚ ← Segmentation, IDS β”‚ β”‚ β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β” β”‚ β”‚ β”‚ β”‚ β”‚ Host β”‚ β”‚ β”‚ ← EDR, Hardening β”‚ β”‚ β”‚ β”Œβ”€β”€β”€β”€β” β”‚ β”‚ β”‚ β”‚ β”‚ β”‚ β”‚App β”‚ β”‚ β”‚ β”‚ ← Auth, Validation β”‚ β”‚ β”‚ β”‚Dataβ”‚ β”‚ β”‚ β”‚ ← Encryption β”‚ β”‚ β”‚ β””β”€β”€β”€β”€β”˜ β”‚ β”‚ β”‚ β”‚ β”‚ β””β”€β”€β”€β”€β”€β”€β”€β”€β”˜ β”‚ β”‚ β”‚ β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜ β”‚ β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜ ``` ## Templates ### Template 1: Mitigation Model ```python from dataclasses import dataclass, field from enum import Enum from typing import List, Dict, Optional, Set from datetime import datetime class ControlType(Enum): PREVENTIVE = "preventive" DETECTIVE = "detective" CORRECTIVE = "corrective" class ControlLayer(Enum): NETWORK = "network" APPLICATION = "application" DATA = "data" ENDPOINT = "endpoint" PROCESS = "process" PHYSICAL = "physical" class ImplementationStatus(Enum): NOT_IMPLEMENTED = "not_implemented" PARTIAL = "partial" IMPLEMENTED = "implemented" VERIFIED = "verified" class Effectiveness(Enum): NONE = 0 LOW = 1 MEDIUM = 2 HIGH = 3 VERY_HIGH = 4 @dataclass class SecurityControl: id: str name: str description: str control_type: ControlType layer: ControlLayer effectiveness: Effectiveness implementation_cost: str # Low, Medium, High maintenance_cost: str status: ImplementationStatus = ImplementationStatus.NOT_IMPLEMENTED mitigates_threats: List[str] = field(default_factory=list) dependencies: List[str] = field(default_factory=list) technologies: List[str] = field(default_factory=list) compliance_refs: List[str] = field(default_factory=list) def coverage_score(self) -> float: """Calculate coverage score based on status and effectiveness.""" status_multiplier = { ImplementationStatus.NOT_IMPLEMENTED: 0.0, ImplementationStatus.PARTIAL: 0.5, ImplementationStatus.IMPLEMENTED: 0.8, ImplementationStatus.VERIFIED: 1.0, } return self.effectiveness.value * status_multiplier[self.status] @dataclass class Threat: id: str name: str category: str # STRIDE category description: str impact: str # Critical, High, Medium, Low likelihood: str risk_score: float @dataclass class MitigationMapping: threat: Threat controls: List[SecurityControl] residual_risk: str = "Unknown" notes: str = "" def calculate_coverage(self) -> float: """Calculate how well controls cover the threat.""" if not self.controls: return 0.0 total_score = sum(c.coverage_score() for c in self.controls) max_possible = len(self.controls) * Effectiveness.VERY_HIGH.value return (total_score / max_possible) * 100 if max_possible > 0 else 0 def has_defense_in_depth(self) -> bool: """Check if multiple layers are covered.""" layers = set(c.layer for c in self.controls if c.status != ImplementationStatus.NOT_IMPLEMENTED) return len(layers) >= 2 def has_control_diversity(self) -> bool: """Check if multiple control types are present.""" types = set(c.control_type for c in self.controls if c.status != ImplementationStatus.NOT_IMPLEMENTED) return len(types) >= 2 @dataclass class MitigationPlan: name: str threats: List[Threat] = field(default_factory=list) controls: List[SecurityControl] = field(default_factory=list) mappings: List[MitigationMapping] = field(default_factory=list) def get_unmapped_threats(self) -> List[Threat]: """Find threats without mitigations.""" mapped_ids = {m.threat.id for m in self.mappings} return [t for t in self.threats if t.id not in mapped_ids] def get_control_coverage(self) -> Dict[str, float]: """Get coverage percentage for each threat.""" return { m.threat.id: m.calculate_coverage() for m in self.mappings } def get_gaps(self) -> List[Dict]: """Identify mitigation gaps.""" gaps = [] for mapping in self.mappings: coverage = mapping.calculate_coverage() if coverage < 50: gaps.append({ "threat": mapping.threat.id, "threat_name": mapping.threat.name, "coverage": coverage, "issue": "Insufficient control coverage", "recommendation": "Add more controls or improve existing ones" }) if not mapping.has_defense_in_depth(): gaps.append({ "threat": mapping.threat.id, "threat_name": mapping.threat.name, "coverage": coverage, "issue": "No defense in depth", "recommendation": "Add controls at different layers" }) if not mapping.has_control_diversity(): gaps.append({ "threat": mapping.threat.id, "threat_name": mapping.threat.name, "coverage": coverage, "issue": "No control diversity", "recommendation": "Add detective/corrective controls" }) return gaps ``` ### Template 2: Control Library ```python class ControlLibrary: """Library of standard security controls.""" STANDARD_CONTROLS = { # Authentication Controls "AUTH-001": SecurityControl( id="AUTH-001", name="Multi-Factor Authentication", description="Require MFA for all user authentication", control_type=ControlType.PREVENTIVE, layer=ControlLayer.APPLICATION, effectiveness=Effectiveness.HIGH, implementation_cost="Medium", maintenance_cost="Low", mitigates_threats=["SPOOFING"], technologies=["TOTP", "WebAuthn", "SMS OTP"], compliance_refs=["PCI-DSS 8.3", "NIST 800-63B"] ), "AUTH-002": SecurityControl( id="AUTH-002", name="Account Lockout Policy", description="Lock accounts after failed authentication attempts", control_type=ControlType.PREVENTIVE, layer=ControlLayer.APPLICATION, effectiveness=Effectiveness.MEDIUM, implementation_cost="Low", maintenance_cost="Low", mitigates_threats=["SPOOFING"], technologies=["Custom implementation"], compliance_refs=["PCI-DSS 8.1.6"] ), # Input Validation Controls "VAL-001": SecurityControl( id="VAL-001", name="Input Validation Framework", description="Validate and sanitize all user input", control_type=ControlType.PREVENTIVE, layer=ControlLayer.APPLICATION, effectiveness=Effectiveness.HIGH, implementation_cost="Medium", maintenance_cost="Medium", mitigates_threats=["TAMPERING", "INJECTION"], technologies=["Joi", "Yup", "Pydantic"], compliance_refs=["OWASP ASVS V5"] ), "VAL-002": SecurityControl( id="VAL-002", name="Web Application Firewall", description="Deploy WAF to filter malicious requests", control_type=ControlType.PREVENTIVE, layer=ControlLayer.NETWORK, effectiveness=Effectiveness.MEDIUM, implementation_cost="Medium", maintenance_cost="Medium", mitigates_threats=["TAMPERING", "INJECTION", "DOS"], technologies=["AWS WAF", "Cloudflare", "ModSecurity"], compliance_refs=["PCI-DSS 6.6"] ), # Encryption Controls "ENC-001": SecurityControl( id="ENC-001", name="Data Encryption at Rest", description="Encrypt sensitive data in storage", control_type=ControlType.PREVENTIVE, layer=ControlLayer.DATA, effectiveness=Effectiveness.HIGH, implementation_cost="Medium", maintenance_cost="Low", mitigates_threats=["INFORMATION_DISCLOSURE"], technologies=["AES-256", "KMS", "HSM"], compliance_refs=["PCI-DSS 3.4", "GDPR Art. 32"] ), "ENC-002": SecurityControl( id="ENC-002", name="TLS Encryption", description="Encrypt data in transit using TLS 1.3", control_type=ControlType.PREVENTIVE, layer=ControlLayer.NETWORK, effectiveness=Effectiveness.HIGH, implementation_cost="Low", maintenance_cost="Low", mitigates_threats=["INFORMATION_DISCLOSURE", "TAMPERING"], technologies=["TLS 1.3", "Certificate management"], compliance_refs=["PCI-DSS 4.1", "HIPAA"] ), # Logging Controls "LOG-001": SecurityControl( id="LOG-001", name="Security Event Logging", description="Log all security-relevant events", control_type=ControlType.DETECTIVE, layer=ControlLayer.APPLICATION, effectiveness=Effectiveness.MEDIUM, implementation_cost="Low", maintenance_cost="Medium", mitigates_threats=["REPUDIATION"], technologies=["ELK Stack", "Splunk", "CloudWatch"], compliance_refs=["PCI-DSS 10.2", "SOC2"] ), "LOG-002": SecurityControl( id="LOG-002", name="Log Integrity Protection", description="Protect logs from tampering", control_type=ControlType.PREVENTIVE, layer=ControlLayer.DATA, effectiveness=Effectiveness.MEDIUM, implementation_cost="Medium", maintenance_cost="Low", mitigates_threats=["REPUDIATION", "TAMPERING"], technologies=["Immutable storage", "Log signing"], compliance_refs=["PCI-DSS 10.5"] ), # Access Control "ACC-001": SecurityControl( id="ACC-001", name="Role-Based Access Control", description="Implement RBAC for authorization", control_type=ControlType.PREVENTIVE, layer=ControlLayer.APPLICATION, effectiveness=Effectiveness.HIGH, implementation_cost="Medium", maintenance_cost="Medium", mitigates_threats=["ELEVATION_OF_PRIVILEGE", "INFORMATION_DISCLOSURE"], technologies=["RBAC", "ABAC", "Policy engines"], compliance_refs=["PCI-DSS 7.1", "SOC2"] ), # Availability Controls "AVL-001": SecurityControl( id="AVL-001", name="Rate Limiting", description="Limit request rates to prevent abuse", control_type=ControlType.PREVENTIVE, layer=ControlLayer.APPLICATION, effectiveness=Effectiveness.MEDIUM, implementation_cost="Low", maintenance_cost="Low", mitigates_threats=["DENIAL_OF_SERVICE"], technologies=["API Gateway", "Redis", "Token bucket"], compliance_refs=["OWASP API Security"] ), "AVL-002": SecurityControl( id="AVL-002", name="DDoS Protection", description="Deploy DDoS mitigation services", control_type=ControlType.PREVENTIVE, layer=ControlLayer.NETWORK, effectiveness=Effectiveness.HIGH, implementation_cost="High", maintenance_cost="Medium", mitigates_threats=["DENIAL_OF_SERVICE"], technologies=["Cloudflare", "AWS Shield", "Akamai"], compliance_refs=["NIST CSF"] ), } def get_controls_for_threat(self, threat_category: str) -> List[SecurityControl]: """Get all controls that mitigate a threat category.""" return [ c for c in self.STANDARD_CONTROLS.values() if threat_category in c.mitigates_threats ] def get_controls_by_layer(self, layer: ControlLayer) -> List[SecurityControl]: """Get controls for a specific layer.""" return [c for c in self.STANDARD_CONTROLS.values() if c.layer == layer] def get_control(self, control_id: str) -> Optional[SecurityControl]: """Get a specific control by ID.""" return self.STANDARD_CONTROLS.get(control_id) def recommend_controls( self, threat: Threat, existing_controls: List[str] ) -> List[SecurityControl]: """Recommend additional controls for a threat.""" available = self.get_controls_for_threat(threat.category) return [c for c in available if c.id not in existing_controls] ``` ### Template 3: Mitigation Analysis ```python class MitigationAnalyzer: """Analyze and optimize mitigation strategies.""" def __init__(self, plan: MitigationPlan, library: ControlLibrary): self.plan = plan self.library = library def calculate_overall_risk_reduction(self) -> float: """Calculate overall risk reduction percentage.""" if not self.plan.mappings: return 0.0 weighted_coverage = 0 total_weight = 0 for mapping in self.plan.mappings: # Weight by threat risk score weight = mapping.threat.risk_score coverage = mapping.calculate_coverage() weighted_coverage += weight * coverage total_weight += weight return weighted_coverage / total_weight if total_weight > 0 else 0 def get_critical_gaps(self) -> List[Dict]: """Find critical gaps that need immediate attention.""" gaps = self.plan.get_gaps() critical_threats = {t.id for t in self.plan.threats if t.impact == "Critical"} return [g for g in gaps if g["threat"] in critical_threats] def optimize_budget( self, budget: float, cost_map: Dict[str, float] ) -> List[SecurityControl]: """Select controls that maximize risk reduction within budget.""" # Simple greedy approach - can be replaced with optimization algorithm recommended = [] remaining_budget = budget unmapped = self.plan.get_unmapped_threats() # Sort controls by effectiveness/cost ratio all_controls = list(self.library.STANDARD_CONTROLS.values()) controls_with_value = [] for control in all_controls: if control.status == ImplementationStatus.NOT_IMPLEMENTED: cost = cost_map.get(control.id, float('inf')) if cost <= remaining_budget: # Calculate value as threats covered * effectiveness / cost threats_covered = len([ t for t in unmapped if t.category in control.mitigates_threats ]) if threats_covered > 0: value = (threats_covered * control.effectiveness.value) / cost controls_with_value.append((control, value, cost)) # Sort by value (higher is better) controls_with_value.sort(key=lambda x: x[1], reverse=True) for control, value, cost in controls_with_value: if cost <= remaining_budget: recommended.append(control) remaining_budget -= cost return recommended def generate_roadmap(self) -> List[Dict]: """Generate implementation roadmap by priority.""" roadmap = [] gaps = self.plan.get_gaps() # Phase 1: Critical threats with low coverage phase1 = [] for gap in gaps: mapping = next( (m for m in self.plan.mappings if m.threat.id == gap["threat"]), None ) if mapping and mapping.threat.impact == "Critical": controls = self.library.get_controls_for_threat(mapping.threat.category) phase1.extend([ { "threat": gap["threat"], "control": c.id, "control_name": c.name, "phase": 1, "priority": "Critical" } for c in controls if c.status == ImplementationStatus.NOT_IMPLEMENTED ]) roadmap.extend(phase1[:5]) # Top 5 for phase 1 # Phase 2: High impact threats phase2 = [] for gap in gaps: mapping = next( (m for m in self.plan.mappings if m.threat.id == gap["threat"]), None ) if mapping and mapping.threat.impact == "High": controls = self.library.get_controls_for_threat(mapping.threat.category) phase2.extend([ { "threat": gap["threat"], "control": c.id, "control_name": c.name, "phase": 2, "priority": "High" } for c in controls if c.status == ImplementationStatus.NOT_IMPLEMENTED ]) roadmap.extend(phase2[:5]) # Top 5 for phase 2 return roadmap def defense_in_depth_analysis(self) -> Dict[str, List[str]]: """Analyze defense in depth coverage.""" layer_coverage = {layer.value: [] for layer in ControlLayer} for mapping in self.plan.mappings: for control in mapping.controls: if control.status in [ImplementationStatus.IMPLEMENTED, ImplementationStatus.VERIFIED]: layer_coverage[control.layer.value].append(control.id) return layer_coverage def generate_report(self) -> str: """Generate comprehensive m
πŸ‘0
πŸ‘οΈ0
πŸ€– Auto-discovered