[teamai] Push 87 resource(s) from XingfenD
This commit is contained in:
@@ -0,0 +1,327 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Check exact terms across final text surfaces without printing matched values."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import codecs
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
import stat
|
||||
import sys
|
||||
import unicodedata
|
||||
|
||||
|
||||
MAX_TERMS_FILE_BYTES = 1024 * 1024
|
||||
MAX_TERMS = 4096
|
||||
MAX_TERM_CHARACTERS = 4096
|
||||
MAX_SURFACE_BYTES = 16 * 1024 * 1024
|
||||
OTHER_DEFAULT_IGNORABLE_RANGES = (
|
||||
(0x034F, 0x034F),
|
||||
(0x115F, 0x1160),
|
||||
(0x17B4, 0x17B5),
|
||||
(0x180B, 0x180F),
|
||||
(0x2065, 0x2065),
|
||||
(0x3164, 0x3164),
|
||||
(0xFE00, 0xFE0F),
|
||||
(0xFFA0, 0xFFA0),
|
||||
(0xFFF0, 0xFFF8),
|
||||
(0x1BCA0, 0x1BCA3),
|
||||
(0x1D173, 0x1D17A),
|
||||
(0xE0000, 0xE0FFF),
|
||||
)
|
||||
|
||||
|
||||
class ScanInputError(ValueError):
|
||||
"""An input cannot be scanned safely."""
|
||||
|
||||
def __init__(self, reason_code: str) -> None:
|
||||
self.reason_code = reason_code
|
||||
super().__init__(reason_code)
|
||||
|
||||
|
||||
def normalize(value: str) -> str:
|
||||
return unicodedata.normalize("NFKC", value).casefold()
|
||||
|
||||
|
||||
def prepare_terms(terms: list[str]) -> list[str]:
|
||||
prepared: list[str] = []
|
||||
seen: set[str] = set()
|
||||
for term in terms:
|
||||
normalized = normalize(term)
|
||||
if normalized and normalized not in seen:
|
||||
seen.add(normalized)
|
||||
prepared.append(normalized)
|
||||
return prepared
|
||||
|
||||
|
||||
def _read_regular_bytes(path: Path, *, maximum_bytes: int) -> bytes:
|
||||
try:
|
||||
entry_stat = path.lstat()
|
||||
except OSError as exc:
|
||||
raise ScanInputError("unreadable_file") from exc
|
||||
if not stat.S_ISREG(entry_stat.st_mode):
|
||||
raise ScanInputError("not_regular_file")
|
||||
if entry_stat.st_size > maximum_bytes:
|
||||
raise ScanInputError("file_too_large")
|
||||
|
||||
flags = os.O_RDONLY | getattr(os, "O_BINARY", 0)
|
||||
flags |= getattr(os, "O_NOFOLLOW", 0)
|
||||
try:
|
||||
descriptor = os.open(path, flags)
|
||||
except OSError as exc:
|
||||
raise ScanInputError("unreadable_file") from exc
|
||||
|
||||
try:
|
||||
opened_stat = os.fstat(descriptor)
|
||||
if not stat.S_ISREG(opened_stat.st_mode):
|
||||
raise ScanInputError("not_regular_file")
|
||||
if (entry_stat.st_dev, entry_stat.st_ino) != (
|
||||
opened_stat.st_dev,
|
||||
opened_stat.st_ino,
|
||||
):
|
||||
raise ScanInputError("file_changed_during_scan")
|
||||
if opened_stat.st_size > maximum_bytes:
|
||||
raise ScanInputError("file_too_large")
|
||||
|
||||
with os.fdopen(descriptor, "rb", closefd=False) as handle:
|
||||
data = handle.read(maximum_bytes + 1)
|
||||
if len(data) > maximum_bytes:
|
||||
raise ScanInputError("file_too_large")
|
||||
return data
|
||||
except OSError as exc:
|
||||
raise ScanInputError("unreadable_file") from exc
|
||||
finally:
|
||||
os.close(descriptor)
|
||||
|
||||
|
||||
def load_terms(path: Path) -> list[str]:
|
||||
data = _read_regular_bytes(path, maximum_bytes=MAX_TERMS_FILE_BYTES)
|
||||
try:
|
||||
text = data.decode("utf-8-sig", errors="strict")
|
||||
except UnicodeError as exc:
|
||||
raise ScanInputError("invalid_terms_encoding") from exc
|
||||
|
||||
raw_terms = [line.strip() for line in text.splitlines()]
|
||||
raw_terms = [term for term in raw_terms if term]
|
||||
if len(raw_terms) > MAX_TERMS:
|
||||
raise ScanInputError("too_many_terms")
|
||||
if any(len(term) > MAX_TERM_CHARACTERS for term in raw_terms):
|
||||
raise ScanInputError("term_too_long")
|
||||
if any(_contains_review_characters(term) for term in raw_terms):
|
||||
raise ScanInputError("unsafe_terms_characters")
|
||||
|
||||
terms = prepare_terms(raw_terms)
|
||||
if not terms:
|
||||
raise ScanInputError("terms_file_empty")
|
||||
return terms
|
||||
|
||||
|
||||
def _matched_prepared_terms(text: str, prepared_terms: list[str]) -> set[int]:
|
||||
normalized = normalize(text)
|
||||
return {index for index, term in enumerate(prepared_terms) if term in normalized}
|
||||
|
||||
|
||||
def count_matches(text: str, terms: list[str]) -> int:
|
||||
return len(_matched_prepared_terms(text, prepare_terms(terms)))
|
||||
|
||||
|
||||
def _decode_surface(data: bytes) -> str:
|
||||
try:
|
||||
if data.startswith(codecs.BOM_UTF32_LE) or data.startswith(codecs.BOM_UTF32_BE):
|
||||
raise ScanInputError("unsupported_text_encoding")
|
||||
if data.startswith(codecs.BOM_UTF16_LE) or data.startswith(codecs.BOM_UTF16_BE):
|
||||
return data.decode("utf-16", errors="strict")
|
||||
return data.decode("utf-8-sig", errors="strict")
|
||||
except UnicodeError as exc:
|
||||
raise ScanInputError("invalid_text_encoding") from exc
|
||||
|
||||
|
||||
def _contains_review_characters(value: str) -> bool:
|
||||
bidi_controls = {
|
||||
"LRE",
|
||||
"RLE",
|
||||
"LRO",
|
||||
"RLO",
|
||||
"PDF",
|
||||
"LRI",
|
||||
"RLI",
|
||||
"FSI",
|
||||
"PDI",
|
||||
"BN",
|
||||
}
|
||||
for character in value:
|
||||
codepoint = ord(character)
|
||||
if unicodedata.category(character) == "Cf":
|
||||
return True
|
||||
if unicodedata.bidirectional(character) in bidi_controls:
|
||||
return True
|
||||
if any(
|
||||
start <= codepoint <= end for start, end in OTHER_DEFAULT_IGNORABLE_RANGES
|
||||
):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _print_payload(payload: dict[str, object]) -> None:
|
||||
print(json.dumps(payload, ensure_ascii=True, sort_keys=True))
|
||||
|
||||
|
||||
def _scan_root(path: Path) -> Path:
|
||||
root_stat = _lstat_for_scan(path)
|
||||
if root_stat is None or not _is_plain_directory(root_stat):
|
||||
raise ScanInputError("scan_root_not_directory")
|
||||
return Path(os.path.abspath(path))
|
||||
|
||||
|
||||
def _lstat_for_scan(path: Path) -> os.stat_result | None:
|
||||
try:
|
||||
return path.lstat()
|
||||
except OSError:
|
||||
return None
|
||||
|
||||
|
||||
def _is_plain_directory(entry_stat: os.stat_result) -> bool:
|
||||
reparse_flag = getattr(stat, "FILE_ATTRIBUTE_REPARSE_POINT", 0)
|
||||
attributes = getattr(entry_stat, "st_file_attributes", 0)
|
||||
return stat.S_ISDIR(entry_stat.st_mode) and not (attributes & reparse_flag)
|
||||
|
||||
|
||||
def _validate_relative_ancestors(root: Path, relative_surface: str) -> None:
|
||||
current = root
|
||||
for component in Path(relative_surface).parts[:-1]:
|
||||
if component in {"", os.curdir, os.pardir}:
|
||||
raise ScanInputError("unsafe_path_component")
|
||||
current /= component
|
||||
entry_stat = _lstat_for_scan(current)
|
||||
if entry_stat is None or not _is_plain_directory(entry_stat):
|
||||
raise ScanInputError("unsafe_path_component")
|
||||
|
||||
|
||||
def _relative_path_surface(path: Path, root: Path) -> str:
|
||||
absolute = os.path.abspath(path)
|
||||
try:
|
||||
common = os.path.commonpath((os.fspath(root), absolute))
|
||||
except ValueError as exc:
|
||||
raise ScanInputError("path_outside_root") from exc
|
||||
if os.path.normcase(common) != os.path.normcase(os.fspath(root)):
|
||||
raise ScanInputError("path_outside_root")
|
||||
relative_surface = Path(os.path.relpath(absolute, root)).as_posix()
|
||||
_validate_relative_ancestors(root, relative_surface)
|
||||
return relative_surface
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Check final text files and filenames for exact forbidden terms."
|
||||
)
|
||||
parser.add_argument("--terms-file", required=True, type=Path)
|
||||
parser.add_argument(
|
||||
"--root",
|
||||
type=Path,
|
||||
help="scan each file's root-relative path, including directory names",
|
||||
)
|
||||
parser.add_argument("paths", nargs="+", type=Path)
|
||||
args = parser.parse_args()
|
||||
|
||||
try:
|
||||
prepared_terms = load_terms(args.terms_file)
|
||||
except ScanInputError as exc:
|
||||
_print_payload({"status": "ERROR", "reason_code": exc.reason_code})
|
||||
return 2
|
||||
|
||||
try:
|
||||
root = _scan_root(args.root) if args.root is not None else None
|
||||
except ScanInputError as exc:
|
||||
_print_payload({"status": "ERROR", "reason_code": exc.reason_code})
|
||||
return 2
|
||||
|
||||
failures: list[dict[str, object]] = []
|
||||
reviews: list[dict[str, object]] = []
|
||||
checked = 0
|
||||
|
||||
for index, path in enumerate(args.paths, start=1):
|
||||
try:
|
||||
path_surface = (
|
||||
_relative_path_surface(path, root) if root is not None else path.name
|
||||
)
|
||||
except ScanInputError as exc:
|
||||
_print_payload(
|
||||
{
|
||||
"status": "ERROR",
|
||||
"files_checked": checked,
|
||||
"file_index": index,
|
||||
"reason_code": exc.reason_code,
|
||||
}
|
||||
)
|
||||
return 2
|
||||
try:
|
||||
data = _read_regular_bytes(path, maximum_bytes=MAX_SURFACE_BYTES)
|
||||
text = _decode_surface(data)
|
||||
except ScanInputError as exc:
|
||||
_print_payload(
|
||||
{
|
||||
"status": "ERROR",
|
||||
"files_checked": checked,
|
||||
"file_index": index,
|
||||
"reason_code": exc.reason_code,
|
||||
}
|
||||
)
|
||||
return 2
|
||||
|
||||
checked += 1
|
||||
content_matches = _matched_prepared_terms(text, prepared_terms)
|
||||
path_matches = _matched_prepared_terms(path_surface, prepared_terms)
|
||||
matched_terms = content_matches | path_matches
|
||||
matched_surfaces: list[str] = []
|
||||
if content_matches:
|
||||
matched_surfaces.append("content")
|
||||
if path_matches:
|
||||
matched_surfaces.append("relative_path" if root is not None else "filename")
|
||||
if matched_terms:
|
||||
failures.append(
|
||||
{
|
||||
"file_index": index,
|
||||
"matched_term_count": len(matched_terms),
|
||||
"surfaces": matched_surfaces,
|
||||
}
|
||||
)
|
||||
|
||||
review_surfaces: list[str] = []
|
||||
if _contains_review_characters(text):
|
||||
review_surfaces.append("content")
|
||||
if _contains_review_characters(path_surface):
|
||||
review_surfaces.append("relative_path" if root is not None else "filename")
|
||||
if review_surfaces:
|
||||
reviews.append(
|
||||
{
|
||||
"file_index": index,
|
||||
"reason_code": "default_ignorable_or_bidi",
|
||||
"surfaces": review_surfaces,
|
||||
}
|
||||
)
|
||||
|
||||
if failures:
|
||||
status = "FAIL"
|
||||
return_code = 1
|
||||
elif reviews:
|
||||
status = "REVIEW"
|
||||
return_code = 1
|
||||
else:
|
||||
status = "PASS"
|
||||
return_code = 0
|
||||
|
||||
_print_payload(
|
||||
{
|
||||
"status": status,
|
||||
"files_checked": checked,
|
||||
"failures": failures,
|
||||
"reviews": reviews,
|
||||
}
|
||||
)
|
||||
return return_code
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user