{"eval":{"version":"openagentskill-skill-eval-v1","slug":"anthropic-webapp-testing","name":"Webapp Testing","generated_at":"2026-07-21T22:43:31.237Z","task_input":"Evaluate Webapp Testing before installing it in an AI agent workflow","status":"review","score":87,"risk_level":"medium","decision":{"recommendation":"manual_review","reason":"Review the audit page, then allow agent install in a sandboxed workflow.","auto_install_allowed":false,"policy":"review","human_review_required":true},"task_fit":{"score":70,"suited_tasks":["Browser automation workflows","Claude Code teams","teams that value GitHub adoption signals","Navigate pages","Click and type safely","Check visual and DOM state","Run test suites","Capture failures"],"suited_agents":["Claude Code","Codex","Cursor","Playwright","OpenAgentSkill CLI","OpenAI Agents","Browser agents","CLI"]},"install":{"command":"npx skills add anthropics/skills --skill webapp-testing","ready":true,"policy":"review","safety_label":"Review before install","targets":[{"id":"openagentskill-cli","label":"CLI","kind":"command","value":"npx skills add anthropics/skills --skill webapp-testing"},{"id":"codex","label":"Codex","kind":"agent-prompt","value":"Install the \"Webapp Testing\" agent skill from https://github.com/anthropics/skills/tree/main/skills/webapp-testing. Read its SKILL.md or equivalent instructions first, install only the files needed for this workspace, and summarize any required setup before using it. Skill purpose: Use Playwright to interact with and test local web applications, capture screenshots, debug UI behavior, and inspect browser logs."},{"id":"claude-code","label":"Claude Code","kind":"agent-prompt","value":"Add \"Webapp Testing\" as a Claude Code skill from https://github.com/anthropics/skills/tree/main/skills/webapp-testing. Inspect the skill instructions, place the reusable skill files in the appropriate local skills location for this project, and report the activation steps. Skill purpose: Use Playwright to interact with and test local web applications, capture screenshots, debug UI behavior, and inspect browser logs."},{"id":"cursor","label":"Cursor","kind":"agent-prompt","value":"Turn \"Webapp Testing\" from https://github.com/anthropics/skills/tree/main/skills/webapp-testing into a reusable Cursor project rule or agent instruction. Preserve the core workflow, adapt paths to this repo, and keep the rule scoped to tasks where it is relevant. Skill purpose: Use Playwright to interact with and test local web applications, capture screenshots, debug UI behavior, and inspect browser logs."}]},"trust":{"score":89,"label":"Production candidate","version":"trust-score-v4","evidence":{"stars":"163K GitHub stars","repoActivity":"163K stars, 19K forks","lastPushed":"4d since push","license":"Source terms (see LICENSE.txt)","repository":"https://github.com/anthropics/skills/tree/main/skills/webapp-testing","install":"npx skills add anthropics/skills --skill webapp-testing","installSafety":"standard package or runtime install path","permissionSurface":"network or browser access","documentation":"Strong README/SKILL.md context","agentOutcomes":"No agent outcome data yet"}},"audit":{"score":93,"risk_level":"safe_to_try","risk_label":"Safe to try","warnings":["Use synthetic or approved test data; do not run destructive browser actions without review."]},"safety_gate":{"score":77,"tier":"reviewed","label":"Reviewed","auto_install_policy":"review","blocked":false,"permission_hints":[{"id":"browser","label":"Browser automation","reason":"Skill may drive a browser or interact with web pages.","severity":"medium"},{"id":"network","label":"Network access","reason":"Skill likely fetches remote pages, APIs, repositories, or external services.","severity":"medium"}],"policy_warnings":["Use synthetic or approved test data; do not run destructive browser actions without review."]},"checks":[{"id":"task_fit","label":"Task fit","status":"warn","score":70,"required_for_auto_install":true,"detail":"Task fit is weak; compare alternatives before selecting.","evidence":["Evaluate Webapp Testing before installing it in an AI agent workflow","browser-automation","Browser automation workflows; Claude Code teams; teams that value GitHub adoption signals"]},{"id":"install_path","label":"Install path","status":"pass","score":92,"required_for_auto_install":true,"detail":"Install handoff is available.","evidence":["npx skills add anthropics/skills --skill webapp-testing"]},{"id":"install_safety","label":"Install command safety","status":"pass","score":92,"required_for_auto_install":true,"detail":"standard package or runtime install path","evidence":["npx skills add anthropics/skills --skill webapp-testing"]},{"id":"trust_score","label":"Trust score","status":"pass","score":89,"required_for_auto_install":true,"detail":"Strong OpenAgentSkill Trust Score across adoption, recent maintenance, license clarity, documentation, dependency/runtime risk, install safety, permission surface, and install availability.","evidence":["Production candidate","163K GitHub stars","Source terms (see LICENSE.txt)"]},{"id":"audit_score","label":"Audit score","status":"pass","score":93,"required_for_auto_install":true,"detail":"Safe to try","evidence":["Use synthetic or approved test data; do not run destructive browser actions without review."]},{"id":"agent_safety_gate","label":"Agent safety gate","status":"warn","score":77,"required_for_auto_install":true,"detail":"Good audit and safety signals with no high-risk permission hints in public metadata.","evidence":["Review the audit page, then allow agent install in a sandboxed workflow.","Safe-to-try audit"]},{"id":"readme_skillmd_completeness","label":"README/SKILL.md completeness","status":"pass","score":90,"required_for_auto_install":false,"detail":"Metadata includes enough usage and workflow context","evidence":["Strong README/SKILL.md context"]},{"id":"license_clarity","label":"License clarity","status":"pass","score":86,"required_for_auto_install":true,"detail":"Source terms (see LICENSE.txt)","evidence":["Source terms (see LICENSE.txt)"]},{"id":"recent_maintenance","label":"Recent maintenance","status":"pass","score":100,"required_for_auto_install":false,"detail":"4d since push","evidence":["4d since push"]},{"id":"permission_surface","label":"Permission surface","status":"pass","score":86,"required_for_auto_install":true,"detail":"network or browser access","evidence":["Browser automation: medium","Network access: medium"]},{"id":"alternatives","label":"Alternatives available","status":"pass","score":82,"required_for_auto_install":false,"detail":"Alternative skills are available for comparison.","evidence":["microsoft-playwright","apify-crawlee","openai-playwright","microsoft-playwright-python"]}],"blockers":[],"warnings":["Task fit: Task fit is weak; compare alternatives before selecting.","Agent safety gate: Good audit and safety signals with no high-risk permission hints in public metadata.","Use synthetic or approved test data; do not run destructive browser actions without review."],"validation_plan":["Inspect repository, README/SKILL.md, license, and recent commits before production use.","Install in an isolated workspace or sandbox with no production secrets available.","Run the smallest representative task and record files touched, commands run, network access, and outputs.","Compare the selected skill against at least one alternative when the eval status is review or failed.","Promote only after the agent reports a successful verification result and unresolved warnings are accepted."],"do_not_use_when":["teams that need a vendor-supported SLA","production agents without a repository review","Use synthetic or approved test data; do not run destructive browser actions without review.","Production credentials, payments, or irreversible account changes without explicit human review","Sensitive private data before reviewing repository code, license, and permission surface"],"alternatives":[{"slug":"microsoft-playwright","name":"Playwright","url":"https://www.openagentskill.com/skills/microsoft-playwright","stars":91270,"install_command":"npx skills add microsoft/playwright","trust_score":91,"audit_score":92},{"slug":"apify-crawlee","name":"Crawlee","url":"https://www.openagentskill.com/skills/apify-crawlee","stars":24036,"install_command":"npx skills add apify/crawlee","trust_score":92,"audit_score":94},{"slug":"openai-playwright","name":"Playwright Browser Skill","url":"https://www.openagentskill.com/skills/openai-playwright","stars":24001,"install_command":"npx skills add openai/skills --skill playwright","trust_score":86,"audit_score":91},{"slug":"microsoft-playwright-python","name":"Playwright Python","url":"https://www.openagentskill.com/skills/microsoft-playwright-python","stars":14833,"install_command":"npx skills add microsoft/playwright-python","trust_score":91,"audit_score":94}],"machine_metadata":{"version":"openagentskill-agent-metadata-v2","skill":{"slug":"anthropic-webapp-testing","name":"Webapp Testing","description":"Use Playwright to interact with and test local web applications, capture screenshots, debug UI behavior, and inspect browser logs.","category":"browser-automation","url":"https://www.openagentskill.com/skills/anthropic-webapp-testing","repository":"https://github.com/anthropics/skills/tree/main/skills/webapp-testing","github_repo":"anthropics/skills"},"suited_tasks":["Browser automation workflows","Claude Code teams","teams that value GitHub adoption signals","Navigate pages","Click and type safely","Check visual and DOM state","Run test suites","Capture failures"],"suited_agents":["Claude Code","Codex","Cursor","Playwright","OpenAgentSkill CLI","OpenAI Agents","Browser agents","CLI"],"install":{"command":"npx skills add anthropics/skills --skill webapp-testing","ready":true,"targets":[{"id":"openagentskill-cli","label":"CLI","kind":"command","value":"npx skills add anthropics/skills --skill webapp-testing"},{"id":"codex","label":"Codex","kind":"agent-prompt","value":"Install the \"Webapp Testing\" agent skill from https://github.com/anthropics/skills/tree/main/skills/webapp-testing. Read its SKILL.md or equivalent instructions first, install only the files needed for this workspace, and summarize any required setup before using it. Skill purpose: Use Playwright to interact with and test local web applications, capture screenshots, debug UI behavior, and inspect browser logs."},{"id":"claude-code","label":"Claude Code","kind":"agent-prompt","value":"Add \"Webapp Testing\" as a Claude Code skill from https://github.com/anthropics/skills/tree/main/skills/webapp-testing. Inspect the skill instructions, place the reusable skill files in the appropriate local skills location for this project, and report the activation steps. Skill purpose: Use Playwright to interact with and test local web applications, capture screenshots, debug UI behavior, and inspect browser logs."},{"id":"cursor","label":"Cursor","kind":"agent-prompt","value":"Turn \"Webapp Testing\" from https://github.com/anthropics/skills/tree/main/skills/webapp-testing into a reusable Cursor project rule or agent instruction. Preserve the core workflow, adapt paths to this repo, and keep the rule scoped to tasks where it is relevant. Skill purpose: Use Playwright to interact with and test local web applications, capture screenshots, debug UI behavior, and inspect browser logs."}],"handoff_url":"https://www.openagentskill.com/api/skills/anthropic-webapp-testing/install","manifest_url":"https://www.openagentskill.com/api/registry/manifest/anthropic-webapp-testing"},"trust":{"score":89,"label":"Production candidate","version":"trust-score-v4","install_policy":"agent_install_candidate","evidence":{"stars":"163K GitHub stars","repoActivity":"163K stars, 19K forks","lastPushed":"4d since push","license":"Source terms (see LICENSE.txt)","repository":"https://github.com/anthropics/skills/tree/main/skills/webapp-testing","install":"npx skills add anthropics/skills --skill webapp-testing","installSafety":"standard package or runtime install path","permissionSurface":"network or browser access","documentation":"Strong README/SKILL.md context","agentOutcomes":"No agent outcome data yet"},"outcome_evidence":{"total":0,"successes":0,"failures":0,"not_relevant":0,"success_rate":null,"recent_success_rate":null,"recent_failure_rate":null,"install_attempts":0,"install_success_rate":null,"risk_blocked":0,"setup_required":0,"avg_output_quality":null,"production_outcomes":0,"last_outcome_at":null,"label":"No agent outcome data yet"},"auto_install":{"allowed":true,"sandbox_required":true,"reason":"Trust Score v4 allows sandbox-first agent installation after normal workspace review."},"best_for":["browser-automation","agent-skill","webapp-testing","playwright","browser","local-testing"],"known_risks":["Use synthetic or approved test data; do not run destructive browser actions without review."]},"agent_proven":{"version":"agent-proven-v1","score":0,"tier":"unproven","label":"Needs first agent run","summary":"No agent outcome reports yet. Use Resolve, run one narrow sandbox task, then report the result.","metrics":{"totalOutcomes":0,"successfulOutcomes":0,"failedOutcomes":0,"installAttempts":0,"installSuccessRate":null,"successRate":null,"recentSuccessRate":null,"recentFailureRate":null,"riskBlocked":0,"setupRequired":0,"notRelevant":0,"avgOutputQuality":null,"avgTimeToUsefulMs":null,"productionOutcomes":0,"humanReviewRequired":0,"uniqueAgents":0,"lastOutcomeAt":null},"signals":[],"penalties":["No real agent outcome evidence yet"]},"audit":{"score":93,"risk_level":"safe_to_try","risk_label":"Safe to try","warnings":["Use synthetic or approved test data; do not run destructive browser actions without review."]},"safety_gate":{"tier":"reviewed","label":"Reviewed","auto_install_policy":"review","auto_install_allowed":false,"human_review_required":true,"blocked":false,"recommended_action":"Review the audit page, then allow agent install in a sandboxed workflow."},"quality":{"score":100,"label":"Excellent"},"supply":{"track":"Coding and developer agents","scenario":"Testing and QA","maintenance":"4d since push","risk":"Safe to try"},"alternative_skills":[{"slug":"microsoft-playwright","name":"Playwright","url":"https://www.openagentskill.com/skills/microsoft-playwright","stars":91270,"install_command":"npx skills add microsoft/playwright","trust_score":91,"audit_score":92},{"slug":"apify-crawlee","name":"Crawlee","url":"https://www.openagentskill.com/skills/apify-crawlee","stars":24036,"install_command":"npx skills add apify/crawlee","trust_score":92,"audit_score":94},{"slug":"openai-playwright","name":"Playwright Browser Skill","url":"https://www.openagentskill.com/skills/openai-playwright","stars":24001,"install_command":"npx skills add openai/skills --skill playwright","trust_score":86,"audit_score":91},{"slug":"microsoft-playwright-python","name":"Playwright Python","url":"https://www.openagentskill.com/skills/microsoft-playwright-python","stars":14833,"install_command":"npx skills add microsoft/playwright-python","trust_score":91,"audit_score":94}],"do_not_use_when":["teams that need a vendor-supported SLA","production agents without a repository review","Use synthetic or approved test data; do not run destructive browser actions without review.","Production credentials, payments, or irreversible account changes without explicit human review","Sensitive private data before reviewing repository code, license, and permission surface"],"agent_contract":{"task_input":"Evaluate Webapp Testing before installing it in an AI agent workflow","recommended_action":"Review the audit page, then allow agent install in a sandboxed workflow.","install_policy":"review","minimum_review_before_use":["Trust: 89/100 Production candidate","Audit: 93/100 Safe to try","Safety: 77/100 Review before install","Review repository, license, install command, and permission surface before production use."],"expected_agent_output":{"selected_skill":"anthropic-webapp-testing (Webapp Testing)","install_command":"npx skills add anthropics/skills --skill webapp-testing","risk_summary":"Safe to try; Reviewed; Low metadata risk","verification_result":"Report the smallest successful task, files touched, warnings, and any missing setup."}},"outcome_feedback":{"endpoint":"https://www.openagentskill.com/api/agent/outcome","method":"POST","requires_resolve_event_id":true,"event_id_source":"Use install_receipt.outcome_feedback.event_id or feedback.event_id returned by /api/agent/resolve for the current task.","expected_outcomes":["success","failed","not_relevant","blocked_by_risk","setup_required"],"payload_template":{"event_id":"<install_receipt.outcome_feedback.event_id or feedback.event_id from /api/agent/resolve>","skill_slug":"anthropic-webapp-testing","task":"Evaluate Webapp Testing before installing it in an AI agent workflow","agent":"codex","outcome":"success","install_used":true,"risk_blocked":false,"setup_required":false,"task_success":true,"output_quality":4,"error_type":null,"human_review_required":false,"workspace":"sandbox","time_to_useful_ms":120000,"notes":"Report the smallest successful task, setup friction, files touched, and risk notes."}},"endpoints":{"web":"https://www.openagentskill.com/skills/anthropic-webapp-testing","api":"https://www.openagentskill.com/api/agent/skills/anthropic-webapp-testing","audit":"https://www.openagentskill.com/skills/anthropic-webapp-testing/audit","eval":"https://www.openagentskill.com/api/agent/evals?slug=anthropic-webapp-testing&task=Evaluate%20Webapp%20Testing%20before%20installing%20it%20in%20an%20AI%20agent%20workflow&max_risk=medium","resolve":"https://www.openagentskill.com/api/agent/resolve?task=Evaluate%20Webapp%20Testing%20before%20installing%20it%20in%20an%20AI%20agent%20workflow&agent=codex&max_risk=medium","receipt":"https://www.openagentskill.com/api/agent/receipt?task=Evaluate%20Webapp%20Testing%20before%20installing%20it%20in%20an%20AI%20agent%20workflow&agent=codex&max_risk=medium&format=text","install":"https://www.openagentskill.com/api/skills/anthropic-webapp-testing/install","manifest":"https://www.openagentskill.com/api/registry/manifest/anthropic-webapp-testing"}},"endpoints":{"web":"https://www.openagentskill.com/skills/anthropic-webapp-testing","api":"https://www.openagentskill.com/api/agent/skills/anthropic-webapp-testing","eval":"https://www.openagentskill.com/api/agent/evals?slug=anthropic-webapp-testing","audit":"https://www.openagentskill.com/skills/anthropic-webapp-testing/audit","resolve":"https://www.openagentskill.com/api/agent/resolve?task=Evaluate%20Webapp%20Testing%20before%20installing%20it%20in%20an%20AI%20agent%20workflow&agent=codex&max_risk=medium"}},"meta":{"endpoint":"/api/agent/evals","mode":"skill_eval","purpose":"Pre-install eval contract for a single skill. Agents should read this before installing a reusable skill.","generated_at":"2026-07-21T22:43:31.237Z"}}