-
Notifications
You must be signed in to change notification settings - Fork 2
Expand file tree
/
Copy pathreport.json
More file actions
168 lines (168 loc) · 12.5 KB
/
Copy pathreport.json
File metadata and controls
168 lines (168 loc) · 12.5 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
{
"schema_version": 1,
"protocol_id": "independent-product-review-v1",
"status": "FAIL",
"status_reason": null,
"host": {
"runner": "codex",
"model": "gpt-5.6-sol",
"provider_family": "openai"
},
"runner_identity": {
"mode": "live",
"path": "/Users/user/.nvm/versions/node/v24.18.0/lib/node_modules/@openai/codex/bin/codex.js",
"sha256": "134063e133f0b4244fa3b251acf973d4fe4b4aeeacbdc135211bf480f59f1477",
"version": "codex-cli 0.146.0"
},
"model_tool_surface": "none",
"source_read_isolation": "prompt-complete-zero-tools",
"credential_environment": "parent-auth-staged-model-tools-disabled",
"execution_mode": "live",
"independent_evidence_eligible": true,
"runner_exit_code": 0,
"elapsed_ms": 135428,
"packet_path": "/private/tmp/e2e-independent-review-20260731/codex/packet.json",
"packet_manifest_path": "/private/tmp/e2e-independent-review-20260731/codex/packet-manifest.json",
"raw_output_path": "/private/tmp/e2e-independent-review-20260731/codex/raw-codex-gpt-5.6-sol.json",
"raw_output_sha256": "5dc14a47e1c1e68c0d4097ce9e6b53d7c67193b912fee740dcf59a28b11917be",
"raw_output_original_sha256": "5dc14a47e1c1e68c0d4097ce9e6b53d7c67193b912fee740dcf59a28b11917be",
"raw_output_exact": true,
"integrity_before": {
"protocol_sha256": "ff5b33d26103c2a21cec91f91a086ad6420188e99f005cef90f77e3932cfd340",
"packet_sha256": "86e95c845a72c020fee66b654045c104db62967d6d30edb5656e4e1e33ed7f26",
"packet_manifest_sha256": "fe8ad5068bdbae2cd06f279a8a5c774d5bc25b48229e56edc9b33622913ae0cf",
"independent_runner_sha256": "de37ef9d00f34355a2a4e69aa84ec3201389d3e1745ab768f5c53574580925a2",
"shared_zero_tool_runner_sha256": "f92c639242752930bd8648d0d7bf4965d4023c0028061e449bf246d73ebf2ab6",
"selected_sources_sha256": "295401a89d9530aedcdeb391ad246d00a0e083c829b28aad4afac5193fb8d543",
"selected_sources": {
"README.md": "8f9721849db78c9fc742db492728f5827aa392d42fdc28e3930e5447951540ac",
"SECURITY.md": "6e03f36f94951f25cf664cebc3f7ee7ae8418a2990d54774d5be784a8020fdcc",
".claude-plugin/plugin.json": "63a37edb76db4b652b15806f866a2cd335a400acf8455cbc91b3e47430098525",
".claude-plugin/marketplace.json": "89006ac66d5d4eb779dcb07d0f82a18cd9b62c8636bd1b914d99122c0c00831a",
".codex-plugin/plugin.json": "a54a828f5b74f8665ff2c7384aac02accfe50643db6e0a9b84031e8dea43fa43",
"skills/playwright-test-generator/SKILL.md": "f7a29ac29549c4d156caec531111251f3410a0a6a9af06a33c7e2b6622dfd4b0",
"skills/e2e-reviewer/SKILL.md": "aa95d44c1a420deff10bbd62138745c0b6758046a96ad9f6385e6b3c21365401",
"skills/playwright-debugger/SKILL.md": "ca789cf06e87fa7ae56236c6e4222d3344783cede7998678b63f84b8061ff6e2",
"skills/cypress-debugger/SKILL.md": "188eb49ffaacd6ad24f058b9fe0709cae767a091449428d4e1316546f49018c0",
"skills/e2e-reviewer/references/pattern-reference.md": "6944980b933e5024ff0a18705cec3a977f6005c8a083b8e2f284ef485a7a5a30",
"skills/e2e-reviewer/references/verification-rules.md": "09f5998f1262d33affcfb44ce417bd52c9a53909336d505fc085526581361559",
"skills/e2e-reviewer/scripts/scan.sh": "1dd555567d8c6a5f0af871137b6c076fee4fd88f5696b2aaf6a3b259ee3d8c22",
"skills/playwright-test-generator/scripts/preflight_target.py": "91b277a46fb885b3f6c2b90d31ac72a61fc10a6bd150abbc03eecbe358e9e822",
"skills/playwright-debugger/scripts/read-playwright-artifact.py": "bf16a2c15a5bd4b9710477fd91b4489f02f3aff55d6f98e25568b88d1b7128f6",
"skills/cypress-debugger/scripts/read-cypress-artifact.py": "baca00a68a81a092b7d19a4d0e0535e6aa273c7fd55caa06c721ecc7c638d0ee",
"skills/cypress-debugger/scripts/extract-junit-failures.py": "9265496da44f17f5e2416f38eb7ff93a30fc7d24c7e6afcc93daf031463dbaa4",
"skills/cypress-debugger/scripts/redact_artifact.py": "2e2c85d6219114817f9a0983c3eaf9bde971c589b043fb1997f45d72e08bad67",
"skills/playwright-test-generator/best-practices.md": "b8472249b1f2406e35775eadfe0e973114436844b1117ec804fb5a09817850db",
"skills/playwright-test-generator/code-rules.md": "7795e6f4cdaa41546c8b5386ce312a9cf67d942cd519d40dd6a7678f03ce5636",
"skills/playwright-test-generator/verification-rules.md": "8564baf915821979263cef7b6e588917f33274d9d8bf7826842f192186378216",
"skills/e2e-reviewer/references/applying-fixes.md": "44e2d1cacbfbe43639dbe28889eea39bd98dac080b161af7244da529a3a8e107",
"skills/e2e-reviewer/references/grep-patterns.md": "decc4fb417f40ea0807eef45d5a5977ce15d49b4b8f896a0ee6114a664dc0e71",
"skills/e2e-reviewer/scripts/parse-ast-grep-json.py": "2c1e2fa2c18a3810495b69491ed0ab74163515fa9936dfae0aaaf73018d37f07",
"scripts/ci/ci-local.sh": "c1b8c715eca3fc4b166fd73eb0a386f7963a2f3870e6d57fb251eb5c373bc46d",
"scripts/ci/pre-push-security.sh": "e3a1703524711461d876a5149185e8654990043eb8897a54dac90cfd97733559"
}
},
"integrity_after": {
"protocol_sha256": "ff5b33d26103c2a21cec91f91a086ad6420188e99f005cef90f77e3932cfd340",
"packet_sha256": "86e95c845a72c020fee66b654045c104db62967d6d30edb5656e4e1e33ed7f26",
"packet_manifest_sha256": "fe8ad5068bdbae2cd06f279a8a5c774d5bc25b48229e56edc9b33622913ae0cf",
"independent_runner_sha256": "de37ef9d00f34355a2a4e69aa84ec3201389d3e1745ab768f5c53574580925a2",
"shared_zero_tool_runner_sha256": "f92c639242752930bd8648d0d7bf4965d4023c0028061e449bf246d73ebf2ab6",
"selected_sources_sha256": "295401a89d9530aedcdeb391ad246d00a0e083c829b28aad4afac5193fb8d543",
"selected_sources": {
"README.md": "8f9721849db78c9fc742db492728f5827aa392d42fdc28e3930e5447951540ac",
"SECURITY.md": "6e03f36f94951f25cf664cebc3f7ee7ae8418a2990d54774d5be784a8020fdcc",
".claude-plugin/plugin.json": "63a37edb76db4b652b15806f866a2cd335a400acf8455cbc91b3e47430098525",
".claude-plugin/marketplace.json": "89006ac66d5d4eb779dcb07d0f82a18cd9b62c8636bd1b914d99122c0c00831a",
".codex-plugin/plugin.json": "a54a828f5b74f8665ff2c7384aac02accfe50643db6e0a9b84031e8dea43fa43",
"skills/playwright-test-generator/SKILL.md": "f7a29ac29549c4d156caec531111251f3410a0a6a9af06a33c7e2b6622dfd4b0",
"skills/e2e-reviewer/SKILL.md": "aa95d44c1a420deff10bbd62138745c0b6758046a96ad9f6385e6b3c21365401",
"skills/playwright-debugger/SKILL.md": "ca789cf06e87fa7ae56236c6e4222d3344783cede7998678b63f84b8061ff6e2",
"skills/cypress-debugger/SKILL.md": "188eb49ffaacd6ad24f058b9fe0709cae767a091449428d4e1316546f49018c0",
"skills/e2e-reviewer/references/pattern-reference.md": "6944980b933e5024ff0a18705cec3a977f6005c8a083b8e2f284ef485a7a5a30",
"skills/e2e-reviewer/references/verification-rules.md": "09f5998f1262d33affcfb44ce417bd52c9a53909336d505fc085526581361559",
"skills/e2e-reviewer/scripts/scan.sh": "1dd555567d8c6a5f0af871137b6c076fee4fd88f5696b2aaf6a3b259ee3d8c22",
"skills/playwright-test-generator/scripts/preflight_target.py": "91b277a46fb885b3f6c2b90d31ac72a61fc10a6bd150abbc03eecbe358e9e822",
"skills/playwright-debugger/scripts/read-playwright-artifact.py": "bf16a2c15a5bd4b9710477fd91b4489f02f3aff55d6f98e25568b88d1b7128f6",
"skills/cypress-debugger/scripts/read-cypress-artifact.py": "baca00a68a81a092b7d19a4d0e0535e6aa273c7fd55caa06c721ecc7c638d0ee",
"skills/cypress-debugger/scripts/extract-junit-failures.py": "9265496da44f17f5e2416f38eb7ff93a30fc7d24c7e6afcc93daf031463dbaa4",
"skills/cypress-debugger/scripts/redact_artifact.py": "2e2c85d6219114817f9a0983c3eaf9bde971c589b043fb1997f45d72e08bad67",
"skills/playwright-test-generator/best-practices.md": "b8472249b1f2406e35775eadfe0e973114436844b1117ec804fb5a09817850db",
"skills/playwright-test-generator/code-rules.md": "7795e6f4cdaa41546c8b5386ce312a9cf67d942cd519d40dd6a7678f03ce5636",
"skills/playwright-test-generator/verification-rules.md": "8564baf915821979263cef7b6e588917f33274d9d8bf7826842f192186378216",
"skills/e2e-reviewer/references/applying-fixes.md": "44e2d1cacbfbe43639dbe28889eea39bd98dac080b161af7244da529a3a8e107",
"skills/e2e-reviewer/references/grep-patterns.md": "decc4fb417f40ea0807eef45d5a5977ce15d49b4b8f896a0ee6114a664dc0e71",
"skills/e2e-reviewer/scripts/parse-ast-grep-json.py": "2c1e2fa2c18a3810495b69491ed0ab74163515fa9936dfae0aaaf73018d37f07",
"scripts/ci/ci-local.sh": "c1b8c715eca3fc4b166fd73eb0a386f7963a2f3870e6d57fb251eb5c373bc46d",
"scripts/ci/pre-push-security.sh": "e3a1703524711461d876a5149185e8654990043eb8897a54dac90cfd97733559"
}
},
"review": {
"summary": "The packet defines extensive framework semantics, fail-closed verification contracts, and trust gates, but three material defects block a strong-quality verdict: a Playwright fix recommends a Jest-DOM matcher, the scanner silently excludes all public directories, and debugger write examples lack an executable path-safety guard.",
"scores": {
"semantic_correctness": 78,
"false_positive_control": 80,
"security_trust_boundaries": 82,
"verification_design": 88,
"scope_contract_consistency": 86,
"docs_usability": 85
},
"findings": [
{
"severity": "H",
"category": "semantic_correctness",
"file": "skills/e2e-reviewer/references/applying-fixes.md",
"line": 29,
"title": "Playwright fix uses a Jest-DOM matcher",
"evidence": "The Playwright replacement table maps getByText(...).toBeTruthy() to toBeInTheDocument(), although Playwright getByText returns a Locator and the packet's authoritative #4f contract instead requires an awaited Playwright locator assertion such as toBeVisible(). Applying this row can generate an unsupported matcher or require unrelated Jest-DOM setup.",
"recommendation": "Move this row exclusively to the RTL/Jest/Vitest section and use await expect(locator).toBeVisible() for proven Playwright Locator subjects."
},
{
"severity": "H",
"category": "false_positive_control",
"file": "skills/e2e-reviewer/scripts/scan.sh",
"line": 1438,
"title": "Blanket public-directory exclusion can produce false-clean scans",
"evidence": "The shared exclusion predicate treats every path under public as scanner-excluded. Consequently valid Playwright or Cypress source placed there is removed from tree validation and all detection tiers, despite the public contracts describing generated/vendor/report/eval exclusions rather than an unconditional public source boundary.",
"recommendation": "Remove the blanket public exclusion, or restrict it to proven generated assets and document the exact exclusion consistently across scanner and public scope contracts."
},
{
"severity": "H",
"category": "security_trust_boundaries",
"file": "skills/playwright-debugger/SKILL.md",
"line": 86,
"title": "Report merge example writes through unchecked paths",
"evidence": "The documented merge command creates playwright-report and redirects output to playwright-report/results.json directly. Earlier prose requires rejecting symlinked roots and destinations, but the packet supplies no executable validation step before this write, so following the example can overwrite a symlink target outside the intended report root.",
"recommendation": "Provide and invoke a bundled descriptor-relative safe-output helper before mkdir and redirection, or have a helper create and atomically publish the validated destination without shell redirection."
}
],
"limitations": [
"No commands or benchmarks were run; scores assess only the frozen contracts and included implementations.",
"Holdouts, raw benchmark evidence, repository history, and approximately 41 KB of source were omitted and were not inferred.",
"Many CI targets referenced by ci-local.sh are absent from the packet, so their implementations and actual outcomes cannot be assessed.",
"The README intentionally excludes several sections, limiting assessment of those public explanations."
],
"verdict": "FAIL"
},
"decision": {
"overall_score": 83.17,
"finding_counts": {
"C": 0,
"H": 3,
"M": 0
},
"checks": {
"overall_score": false,
"dimension_floor": false,
"critical_findings": true,
"high_findings": false,
"model_verdict_matches": true
}
},
"limitations": [
"This is independent context/model-family evidence, not human review.",
"The public packet is not a sealed benchmark or independent ground truth.",
"The Anthropic entries are two models in one provider family, not two independent providers.",
"Excluded scorecards, holdouts, reports, prior reviews, chat conclusions, and git history reduce but cannot prove absence of training-data contamination."
]
}