Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -143,7 +143,7 @@ jobs:
- name: Run CI Runtime Sanitizer Suite
run: |
ccache -z
LLM_EDGEFLOW_LINKER=mold LLM_EDGEFLOW_JOBS=4 LLM_EDGEFLOW_SANITIZERS=address,undefined \
LLM_EDGEFLOW_LINKER=mold LLM_EDGEFLOW_SANITIZERS=address,undefined \
./scripts/run_sanitizers.sh --ci-runtime
ccache -s

Expand Down
4 changes: 3 additions & 1 deletion CONTRIBUTING.md
Original file line number Diff line number Diff line change
Expand Up @@ -33,7 +33,9 @@ working branch or merge other working branches into it. If main advances, explic
the branch onto the latest main or recreate it there with only the current PR's changes, then
revalidate. Published history rewrites require explicit authorization and coordination with
other users of the branch. The delivery script checks this history before the local gate and
again before an authorized merge; it never rebases automatically.
again before an authorized merge; it never rebases automatically. Before merging, the remote
PR head must match the locally verified commit; the script validates that exact SHA and binds
the merge to it so a concurrent branch update stops delivery.

The branch itself is not evidence of quality; it provides isolation and a reviewable diff.

Expand Down
11 changes: 11 additions & 0 deletions dev_support/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -17,3 +17,14 @@ runner compiles these examples directly. No starter file is linked into a produc

`benchmarks/` and `node_authoring/benchmark/` contain opt-in development measurements. They are
separate from the default runtime and correctness suites.

After running `./scripts/run_all_tests.sh`, measure the current template and rule snapshot
Nodes, with and without concurrent Control updates, using an empty output directory:

```bash
python3 dev_support/benchmarks/control_snapshots.py --output-dir /tmp/edgeflow-control-snapshots --rounds 1
```

The script saves request timing, Control allocation counts, command logs and median summaries
in that directory. It compiles only current sources; historical source comparisons and the
`--baseline` option have been removed.
2 changes: 1 addition & 1 deletion dev_support/benchmarks/control_snapshots.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -113,7 +113,7 @@ int main(int argc, char** argv) {
} else {
auto* out = ctx.Read<RuleMatchBatch>("matches");
if (!out || out->size() != 50 || (*out)[0].data.category != "GREETING" ||
(*out)[0].data.captures.at("tail") != "world")
(*out)[0].data.slots.at("tail") != "world")
return 6;
}
}
Expand Down
78 changes: 34 additions & 44 deletions dev_support/benchmarks/control_snapshots.py
Original file line number Diff line number Diff line change
@@ -1,11 +1,11 @@
#!/usr/bin/env python3
"""Compare configuration snapshot node implementations on an otherwise idle machine.
"""Measure current configuration snapshot nodes on an otherwise idle machine.

Run the canonical gate/build first. This script reuses its Ninja node-runner
runtime objects and libraries without building the repository. Only the two node
translation units are replaced for the baseline; this is not a full historical
checkout benchmark. Both versions use C++17, -O3 and -DNDEBUG. Each invocation
processes 2,000 requests of 50 samples; writer updates are spaced by 100 us.
runtime objects and libraries without building the repository. The benchmark
and the two current node translation units use C++17, -O3 and -DNDEBUG; historical
sources are not compiled. Each invocation processes 2,000 requests of 50 samples;
writer updates are spaced by 100 us.
Ordinary C++ allocation counts are measured separately from request timing.
"""

Expand Down Expand Up @@ -60,7 +60,6 @@ def main():
parser.add_argument("--build-dir", type=Path, default=ROOT / "build")
parser.add_argument("--output-dir", type=Path, required=True,
help="empty temporary directory for binaries and all evidence")
parser.add_argument("--baseline", default="7a6ca02")
parser.add_argument("--rounds", type=positive_integer, default=7)
args = parser.parse_args()
build = args.build_dir.resolve()
Expand Down Expand Up @@ -91,7 +90,6 @@ def run(command, cwd=ROOT):
environment = "".join(run(command) for command in (
["uname", "-a"], ["c++", "--version"], ["lscpu"],
["git", "rev-parse", "HEAD"], ["git", "status", "--short"],
["git", "rev-parse", args.baseline],
))
environment += f"\narguments: {vars(args)}\n"
(output / "environment.txt").write_text(environment)
Expand All @@ -105,54 +103,46 @@ def run(command, cwd=ROOT):
))
benchmark_object = output / "bench.o"
run(flags + ["-c", Path(__file__).with_suffix(".cpp"), "-o", benchmark_object])
for version in ("baseline", "current"):
objects = []
for node in NODES:
relative_source = f"src/common_nodes/{node}.cpp"
source = output / f"{version}_{node}.cpp"
source.write_text(
run(["git", "show", f"{args.baseline}:{relative_source}"])
if version == "baseline" else (ROOT / relative_source).read_text()
)
obj = output / f"{version}_{node}.o"
run(flags + ["-c", source, "-o", obj])
objects.append(str(obj))
command = [token for token in link if not any(
token.endswith(f"/{node}.cpp.o") for node in NODES
)]
command[command.index("-o") + 1] = str(output / f"bench_{version}")
# Put replacement objects ahead of static libraries for normal linkers.
command[command.index("-o"):command.index("-o")] = objects + [str(benchmark_object)]
run(command, cwd=build)
objects = []
for node in NODES:
source = ROOT / f"src/common_nodes/{node}.cpp"
obj = output / f"{node}.o"
run(flags + ["-c", source, "-o", obj])
objects.append(str(obj))
command = [token for token in link if not any(
token.endswith(f"/{node}.cpp.o") for node in NODES
)]
executable = output / "bench_current"
command[command.index("-o") + 1] = str(executable)
# Put replacement objects ahead of static libraries for normal linkers.
command[command.index("-o"):command.index("-o")] = objects + [str(benchmark_object)]
run(command, cwd=build)

records = []
for round_index in range(args.rounds):
for node in ("template", "rules"):
for concurrent in (0, 1):
versions = ("baseline", "current") if round_index % 2 == 0 else ("current", "baseline")
for version in versions:
stdout = run([output / f"bench_{version}", node, concurrent])
(output / f"{round_index}_{node}_{concurrent}_{version}.log").write_text(stdout)
result = next(line.split() for line in stdout.splitlines() if line.startswith("RESULT "))
allocation = next(line.split() for line in stdout.splitlines() if line.startswith("ALLOC "))
records.append(dict(
round=round_index, node=node, concurrent=concurrent, version=version,
us=float(result[3]), updates=int(result[4]),
allocations=int(allocation[1]), bytes=int(allocation[2]),
))
(output / "results.json").write_text(json.dumps(records, indent=2) + "\n")
stdout = run([executable, node, concurrent])
(output / f"{round_index}_{node}_{concurrent}.log").write_text(stdout)
result = next(line.split() for line in stdout.splitlines() if line.startswith("RESULT "))
allocation = next(line.split() for line in stdout.splitlines() if line.startswith("ALLOC "))
records.append(dict(
round=round_index, node=node, concurrent=concurrent,
us=float(result[3]), updates=int(result[4]),
allocations=int(allocation[1]), bytes=int(allocation[2]),
))
(output / "results.json").write_text(json.dumps(records, indent=2) + "\n")
print(f"Completed round {round_index + 1}/{args.rounds}", flush=True)

summary = []
for node in ("template", "rules"):
for concurrent in (0, 1):
medians = {version: statistics.median(
record["us"] for record in records
selected = [record for record in records
if record["node"] == node and record["concurrent"] == concurrent
and record["version"] == version
) for version in ("baseline", "current")}
change = 100 * (medians["current"] / medians["baseline"] - 1)
summary.append(f"{node} concurrent={concurrent}: {medians}, change_pct={change}\n")
]
medians = {key: statistics.median(record[key] for record in selected)
for key in ("us", "allocations", "bytes")}
summary.append(f"{node} concurrent={concurrent}: medians={medians}\n")
(output / "summary.txt").write_text("".join(summary))
print("".join(summary), end="")

Expand Down
6 changes: 6 additions & 0 deletions doc/CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,12 @@

## Unreleased

开发工具修复:Markdown 链接检查支持单引号与圆括号标题、带空格的尖括号目标及平衡或转义的
目标圆括号;生产版 `alg_pipeline_tool edit` 根据 `validation.diagnostics` 提示构建变体与测试工具。
交付脚本要求远端 PR head 与已验证提交一致,以该 SHA 复核历史,并通过 `--match-head-commit`
阻止检查后的并发更新进入合并。Control 快照基准改用当前 `slots` DTO,仅测量当前源码,删除历史
源码对比和 `--baseline` 参数。

架构审查回归修复:Map 的回调通过移动交给运行时,支持捕获 `unique_ptr` 等不可复制状态;
流契约错误由 Validator 提供生产者、消费者、有效端口契约与推导出的数量形状,CLI 据此解释
节点输入、业务出口及 IO 边界错误,保留拆分来源并正确区分逐项配对与出口数量要求。
Expand Down
88 changes: 84 additions & 4 deletions scripts/check_doc_links.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,14 +4,17 @@
import argparse
from pathlib import Path
import re
import string
import subprocess
import tempfile
import unicodedata
from urllib.parse import unquote


ATX_HEADING = re.compile(r"^(#{1,6})\s+(.*?)\s*#*\s*$")
INLINE_LINK = re.compile(r"\]\(\s*<?([^)\s>]+)>?(?:\s+\"[^\"]*\")?\s*\)")
INLINE_LINK_START = re.compile(r"\]\(")
INLINE_LINK_END = re.compile(
r'''(?:[ \t]+(?:"(?:\\.|[^"\\])*"|'(?:\\.|[^'\\])*'|\((?:\\.|[^()\\])*\)))?[ \t]*\)''')
HTML_TARGET = re.compile(r"\b(?:href|src)=\"([^\"]+)\"")
HTML_ANCHOR = re.compile(r"<a\s+(?:id|name)=\"([^\"]+)\"")
INLINE_CODE = re.compile(r"`[^`]*`")
Expand All @@ -20,6 +23,56 @@
WALK_EXCLUDES = {".git", "3rdparty", "results", "output", "Testing"}


def link_destination(text, position):
"""Read an angle-delimited or balanced bare destination and its end offset."""
angle_delimited = text[position:position + 1] == "<"
if angle_delimited:
position += 1
destination, depth = [], 0
while position < len(text):
char = text[position]
if char == "\\" and position + 1 < len(text) and text[position + 1] in string.punctuation:
destination.append(text[position + 1])
position += 2
continue
if angle_delimited:
if char == ">":
return "".join(destination), position + 1
if char in "<\r\n":
return None
elif char <= " " or char == "\x7f":
break
elif char == "(":
depth += 1
elif char == ")":
if depth == 0:
break
depth -= 1
destination.append(char)
position += 1
if angle_delimited or depth:
return None
return "".join(destination), position


def inline_link_targets(text):
"""Yield single-line inline destinations, excluding optional link titles."""
position = 0
while match := INLINE_LINK_START.search(text, position):
position = match.end()
start = position
while start < len(text) and text[start] in " \t":
start += 1
parsed = link_destination(text, start)
if parsed is None:
continue
destination, end = parsed
suffix = INLINE_LINK_END.match(text, end)
if suffix:
position = suffix.end()
yield destination


def slugify(heading):
"""GitHub heading anchor: lowercase, drop punctuation/symbols, spaces to '-'."""
text = re.sub(r"`([^`]*)`", r"\1", heading)
Expand Down Expand Up @@ -90,7 +143,7 @@ def check(root):
relative = path.relative_to(root)
for number, line in prose_lines(path):
prose = INLINE_CODE.sub("", line)
for target in INLINE_LINK.findall(prose) + HTML_TARGET.findall(prose):
for target in [*inline_link_targets(prose), *HTML_TARGET.findall(prose)]:
if URL_SCHEME.match(target) or target.startswith("//"):
continue
counters["links"] += 1
Expand Down Expand Up @@ -144,12 +197,39 @@ def write(root, name, text):
# Seven inline links and one img src; the https link is skipped.
assert counters == {"files": 2, "links": 8, "anchor_links": 4}, counters

write(root, "broken.md", "[x](guide/missing.md) [y](guide/target.md#nope) [z](../outside.md)\n")
write(root, "guide/guide_(old).md", "# Heading\n")
write(root, "guide/guide_(old_(nested_(v1))).md", "# Heading\n")
write(root, "guide/file space.md", "# Heading\n")
valid_links = [
'[x](guide/guide_(old).md#heading)',
'[x](guide/guide_(old_(nested_(v1))).md#heading)',
r'[x](guide/guide_\(old\).md#heading)',
"[x](<guide/file space.md#heading> 'caption')",
"[x](guide/target.md 'caption')",
'[x](guide/target.md (caption))',
r'''[x](guide/target.md "say \"hello\"")''',
'[x](guide/target.md "title [skip](missing.md)")',
]
write(root, "syntax.md", "\n".join(valid_links) + "\n"
"[literal](guide_(unbalanced.md)\n"
"[literal](<guide/target.md>'missing separator')\n"
"[literal](guide/target.md (nested(title)))\n")
errors, counters = check(root)
assert errors == [], errors
assert counters["links"] == 8 + len(valid_links), counters

write(root, "broken.md",
"[x](guide/missing.md) [y](guide/target.md#nope) [z](../outside.md)\n"
"[single](missing.md 'caption') [paren](missing.md (caption))\n"
"[balanced](missing_(old).md) [angle](<missing file.md> 'caption')\n")
errors, _ = check(root)
assert len(errors) == 3, errors
assert len(errors) == 7, errors
assert "missing link target" in errors[0], errors
assert "missing heading anchor" in errors[1], errors
assert "leaves the repository" in errors[2], errors
for error, target in zip(errors[3:],
("missing.md", "missing.md", "missing_(old).md", "missing file.md")):
assert error.endswith("missing link target: " + target), errors

with tempfile.TemporaryDirectory(prefix="doc-links-empty-") as directory:
errors, _ = check(Path(directory))
Expand Down
20 changes: 16 additions & 4 deletions scripts/git_branch_upload.sh
Original file line number Diff line number Diff line change
Expand Up @@ -65,15 +65,16 @@ case "${BRANCH_NAME}" in
esac

verify_branch_history() {
local checked_head="${1:-HEAD}"
git fetch origin main
if ! git merge-base --is-ancestor origin/main HEAD; then
if ! git merge-base --is-ancestor origin/main "${checked_head}"; then
echo "Error: origin/main is not an ancestor of ${BRANCH_NAME}."
echo "Explicitly rebase onto origin/main or recreate the branch there with only this PR's changes."
echo "Do not merge main into the working branch; this script will not rewrite history."
exit 1
fi
local branch_merges
branch_merges="$(git rev-list --merges origin/main..HEAD)"
branch_merges="$(git rev-list --merges "origin/main..${checked_head}")"
if [[ -n "${branch_merges}" ]]; then
echo "Error: ${BRANCH_NAME} contains merge commits above origin/main."
echo "Recreate or explicitly rebase the branch with only this PR's linear commits, then rerun."
Expand Down Expand Up @@ -107,6 +108,7 @@ if git diff --quiet origin/main...HEAD; then
fi

echo "[4/5] Pushing branch and creating or reusing its PR..."
VERIFIED_HEAD="$(git rev-parse HEAD)"
git push -u origin "${BRANCH_NAME}"
if ! gh pr view "${BRANCH_NAME}" --json number >/dev/null 2>&1; then
gh pr create \
Expand Down Expand Up @@ -140,9 +142,19 @@ if [[ "${DELIVERY_MODE}" == "--pr-only" ]]; then
fi

PR_NUMBER="$(gh pr view "${BRANCH_NAME}" --json number --jq '.number')"
if ! PR_HEAD_SHA="$(gh pr view "${PR_NUMBER}" --json headRefOid --jq '.headRefOid')" || \
[[ ! "${PR_HEAD_SHA}" =~ ^[0-9a-fA-F]{40}$ ]]; then
echo "Error: cannot confirm the head SHA of PR #${PR_NUMBER}; merge was not attempted."
exit 1
fi
if [[ "${PR_HEAD_SHA}" != "${VERIFIED_HEAD}" ]]; then
echo "Error: PR #${PR_NUMBER} head changed since the verified branch was pushed."
echo "Fetch the updated branch and rerun verification before merging."
exit 1
fi
echo "Rechecking branch history against the latest origin/main before merge..."
verify_branch_history
gh pr merge "${PR_NUMBER}" --merge --delete-branch
verify_branch_history "${PR_HEAD_SHA}"
gh pr merge "${PR_NUMBER}" --merge --delete-branch --match-head-commit "${PR_HEAD_SHA}"
if ! MERGE_SHA="$(gh pr view "${PR_NUMBER}" --json mergeCommit --jq '.mergeCommit.oid // empty')" || \
[[ -z "${MERGE_SHA}" ]]; then
echo "Error: cannot confirm the merge SHA of PR #${PR_NUMBER}; main CI was not verified."
Expand Down
21 changes: 16 additions & 5 deletions src/cli/alg_pipeline_tool.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -67,12 +67,21 @@ void PrintRegistrationHint(const nlohmann::json& response,
auto unknown_registration = [](const std::string& code) {
return code == "UNKNOWN_MODEL_TYPE" || code == "UNKNOWN_BACKEND";
};
bool needs_hint = unknown_registration(source_code);
auto diagnostics = response.find("diagnostics");
if (diagnostics != response.end() && diagnostics->is_array()) {
auto has_unknown_registration = [&](const nlohmann::json& report) {
auto diagnostics = report.find("diagnostics");
if (diagnostics == report.end() || !diagnostics->is_array()) return false;
for (const auto& diagnostic : *diagnostics) {
if (unknown_registration(diagnostic.value("code", ""))) needs_hint = true;
if (diagnostic.is_object() &&
unknown_registration(diagnostic.value("code", "")))
return true;
}
return false;
};
bool needs_hint =
unknown_registration(source_code) || has_unknown_registration(response);
auto validation = response.find("validation");
if (validation != response.end() && validation->is_object()) {
needs_hint = needs_hint || has_unknown_registration(*validation);
}
if (needs_hint) {
std::cerr << "提示:当前 alg_pipeline_tool 只包含本次构建启用的生产注册。\n"
Expand Down Expand Up @@ -717,7 +726,9 @@ int main(int argc, char* argv[]) {
}
try {
auto result = llm_edgeflow::PipelineAuthoring::ApplyRequest(request);
std::cout << result.ToJson().dump(2) << std::endl;
auto response = result.ToJson();
std::cout << response.dump(2) << std::endl;
PrintRegistrationHint(response);
return result.ok ? 0 : 1;
} catch (const std::exception& error) {
std::cout << ToolError("AUTHORING_ERROR", error.what()).dump(2)
Expand Down
Loading
Loading