fix: address code scanning findings

This commit is contained in:
xixu-me committed 2026-06-14 19:47:40 +08:00
1 parent 4256bb76a7
commit b06168dadd
2 files changed
+21 -4

No files matched your search

@@ -1,6 +1,7 @@
import json import json
import re import re
from pathlib import Path from pathlib import Path
from urllib.parse import urlparse
KB_ROOT = Path(__file__).resolve().parents[1] KB_ROOT = Path(__file__).resolve().parents[1]
@@ -16,6 +17,7 @@ PROMPT_INJECTION_RE = re.compile(
r"(ignore previous instructions|disregard previous instructions|reveal the system prompt)", r"(ignore previous instructions|disregard previous instructions|reveal the system prompt)",
re.IGNORECASE, re.IGNORECASE,
) )
URL_RE = re.compile(r"https?://[^\s<>)\]]+")
VISIBLE_FORBIDDEN_PATTERNS = ( VISIBLE_FORBIDDEN_PATTERNS = (
"../private", "../private",
@@ -98,6 +100,14 @@ def rel(path: Path) -> str:
return path.relative_to(KB_ROOT).as_posix() return path.relative_to(KB_ROOT).as_posix()
def contains_insecure_docs_python_url(text: str) -> bool:
for match in URL_RE.finditer(text):
parsed = urlparse(match.group(0).rstrip(".,;:"))
if parsed.scheme == "http" and parsed.netloc == "docs.python.org":
return True
return False
def frontmatter_value(text: str, key: str) -> str | None: def frontmatter_value(text: str, key: str) -> str | None:
lines = text.splitlines() lines = text.splitlines()
if not lines or lines[0] != "---": if not lines or lines[0] != "---":
@@ -285,7 +295,7 @@ def test_semantic_lint_known_findings_stay_fixed():
errors.append("summaries/01_Python.md does not clarify Python 3.6 historical context") errors.append("summaries/01_Python.md does not clarify Python 3.6 historical context")
if "不保证 URL 长期可用" not in python_summary or "本地 XML 示例文件" not in python_summary: if "不保证 URL 长期可用" not in python_summary or "本地 XML 示例文件" not in python_summary:
errors.append("summaries/01_Python.md does not clarify external API example stability") errors.append("summaries/01_Python.md does not clarify external API example stability")
if "http://docs.python.org" in python_summary: if contains_insecure_docs_python_url(python_summary):
errors.append("summaries/01_Python.md still uses HTTP docs.python.org link") errors.append("summaries/01_Python.md still uses HTTP docs.python.org link")
if "[[summaries/practical-python-attribution]]" not in overview: if "[[summaries/practical-python-attribution]]" not in overview:
@@ -321,7 +331,7 @@ def test_semantic_lint_known_findings_stay_fixed():
WIKI_ROOT / "concepts" / "Python-文档与帮助系统.md", WIKI_ROOT / "concepts" / "Python-文档与帮助系统.md",
WIKI_ROOT / "exercises" / "1-2-getting-help.md", WIKI_ROOT / "exercises" / "1-2-getting-help.md",
]: ]:
if "http://docs.python.org" in read_text(path): if contains_insecure_docs_python_url(read_text(path)):
errors.append(f"{rel(path)} still uses HTTP docs.python.org link") errors.append(f"{rel(path)} still uses HTTP docs.python.org link")
for target in REMOVED_CONCEPT_ALIAS_TARGETS: for target in REMOVED_CONCEPT_ALIAS_TARGETS:
@@ -341,6 +351,12 @@ def test_semantic_lint_known_findings_stay_fixed():
assert errors == [] assert errors == []
def test_docs_python_http_detection_matches_exact_host():
assert contains_insecure_docs_python_url("See http://docs.python.org/3/library/pathlib.html")
assert not contains_insecure_docs_python_url("See https://docs.python.org/3/library/pathlib.html")
assert not contains_insecure_docs_python_url("See http://docs.python.org.example.com/3/library/pathlib.html")
def test_followup_enhancements_are_indexed_and_scoped(): def test_followup_enhancements_are_indexed_and_scoped():
index_text = read_text(WIKI_ROOT / "index.md") index_text = read_text(WIKI_ROOT / "index.md")
errors = [] errors = []
+3 -2
View File
@@ -8,6 +8,7 @@ import re
import subprocess import subprocess
import sys import sys
import time import time
import uuid
from datetime import datetime, timezone from datetime import datetime, timezone
from pathlib import Path from pathlib import Path
from typing import Any, Iterable from typing import Any, Iterable
@@ -98,6 +99,7 @@ REQUIRED_JOURNEY_EVENT_KEYS = {
"visible_state", "visible_state",
"api_state", "api_state",
} }
FINGERPRINT_NAMESPACE = uuid.UUID("46ad4a87-ef03-4e26-a1a7-a7e5f9c66a93")
class DiscoveryLoop: class DiscoveryLoop:
@@ -372,7 +374,6 @@ class DiscoveryLoop:
"message_present": bool(parsed.get("message")), "message_present": bool(parsed.get("message")),
"expected_state": parsed.get("expected_state"), "expected_state": parsed.get("expected_state"),
"validation": validation, "validation": validation,
"actor_prompt_hash": hashlib.sha256(actor_prompt.encode("utf-8")).hexdigest()[:16],
"model_metadata": self.actor_model_metadata(), "model_metadata": self.actor_model_metadata(),
"untrusted_evidence": True, "untrusted_evidence": True,
} }
@@ -2896,7 +2897,7 @@ def fingerprint_finding(finding: dict[str, Any], policy: dict[str, Any]) -> str:
fields = [str(item) for item in policy["issueClustering"]["fingerprintFields"]] fields = [str(item) for item in policy["issueClustering"]["fingerprintFields"]]
payload = {field: finding.get(field) for field in fields} payload = {field: finding.get(field) for field in fields}
encoded = json.dumps(payload, sort_keys=True, ensure_ascii=False) encoded = json.dumps(payload, sort_keys=True, ensure_ascii=False)
return hashlib.sha256(encoded.encode("utf-8")).hexdigest()[:24] return uuid.uuid5(FINGERPRINT_NAMESPACE, encoded).hex[:24]
def build_issue_clusters(findings: list[dict[str, Any]], policy: dict[str, Any]) -> list[dict[str, Any]]: def build_issue_clusters(findings: list[dict[str, Any]], policy: dict[str, Any]) -> list[dict[str, Any]]: