Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .claude-plugin/plugin.json
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
{
"name": "autocode",
"version": "3.0.0",
"version": "3.1.0",
"description": "Claude Code plugin for verified competitive-programming problem-setting workflows.",
"author": {
"name": "SummerOneTwo",
Expand Down
2 changes: 1 addition & 1 deletion .codex-plugin/plugin.json
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
{
"name": "autocode",
"version": "3.0.0",
"version": "3.1.0",
"description": "Verified competitive-programming problem authoring workflows for Claude Code and Codex.",
"author": {
"name": "SummerOneTwo",
Expand Down
7 changes: 7 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -7,8 +7,15 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

## [3.1.0] - 2026-09-19

### Added

- **多规模阶梯数据采样与经验复杂度拟合**:
- 新增 `MultiScaleSampler`(多规模阶梯数据采样器),支持自适应 5 点阶梯采样,适配多项式、小规模多项式与指数级规模,杜绝采样点倒挂;支持 testlib 命令行规范参数与多测极端数据分布生成。
- 新增 `DynamicExecutionMonitor`(动态执行监控器),采集纯 CPU 耗时(`utime + stime`)与物理内存,测量并扣除原生系统启动底噪;支持交互题双向匿名管道并发调度与独立 CPU 耗时核算;完善进程树深度回收与异常清理。
- 新增 `EmpiricalRatioAnalyzer`(经验倍率拟合分析器),采用对数线性回归拟合幂指数 $\alpha$ 与判定系数 $R^2$;引入动态理论期望倍率与自适应容差校验;支持处理器缓存容量跨越保护(Cache Jump Protection);支持阶乘复杂度 $O(n!)$ 校验。
- 重构 `complexity`、`solution_audit` 与 `audit` 工具层:解法审计与全量审计接入多规模经验拟合结果与质量信号门禁,未通过时追加高优先级阻断与修复指引。
- **DeepSeek Harness (DSH) 插件生态支持**:
- 新增 `.dsh-plugin/package.json`、`.dsh-plugin/cordis.patch.yml`、`.dsh-plugin/index.js`,支持在 DSH 会话中挂载 AutoCode 作为 MCP 服务。
- 通过 `dsh-agent-instructions` 扩展机制将工作区指导规范与质量门禁自动注入大语言模型系统提示词上下文。
Expand Down
2 changes: 1 addition & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
[project]
name = "autocode-mcp"
version = "3.0.0"
version = "3.1.0"
description = "MCP Server for competitive programming problem creation, based on AutoCode paper"
readme = "README.md"
requires-python = ">=3.10"
Expand Down
2 changes: 1 addition & 1 deletion src/autocode_mcp/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@
"""
import os

__version__ = "3.0.0"
__version__ = "3.1.0"

# 获取 templates 目录路径(包内目录)
_PACKAGE_DIR = os.path.dirname(__file__)
Expand Down
91 changes: 82 additions & 9 deletions src/autocode_mcp/tools/audit.py
Original file line number Diff line number Diff line change
@@ -1,16 +1,9 @@
"""High-level problem audit tool.

This tool aggregates deterministic evidence from the AutoCode problem package.
It does not call an LLM; the LLM-facing difficulty explanation can consume the
returned signals.
"""

from __future__ import annotations

import json
from datetime import datetime, timezone
from pathlib import Path
from typing import Literal
from typing import Any, Literal

from pydantic import ValidationError

Expand All @@ -19,7 +12,12 @@
from ..workflow.guard import signal_satisfied as _guard_signal_satisfied
from ..workflow.models import AutoCodeManifest
from .base import Tool, ToolResult, input_schema_from_model
from .complexity import analyze_loop_complexity, detect_algorithm_patterns
from .complexity import (
analyze_loop_complexity,
detect_algorithm_patterns,
extract_claimed_complexity,
run_empirical_verification,
)
from .schemas import ProblemAuditInput
from .test_verify import ProblemVerifyTestsTool

Expand Down Expand Up @@ -120,6 +118,37 @@ async def execute(
blocking=blocking,
next_actions=next_actions,
)
empirical_signal = await self._empirical_complexity_signal(problem_path, manifest)
quality_signals["empirical_complexity"] = empirical_signal
if empirical_signal.get("executed") and not empirical_signal.get("passed"):
evidence = empirical_signal.get("evidence")
failure_reason = (
str(evidence.get("failure_reason", "empirical complexity verification failed"))
if isinstance(evidence, dict)
else "empirical complexity verification failed"
)
blocking.append(
{
"gate": "empirical_complexity",
"reason": failure_reason,
}
)
next_actions.append(
{
"tool_name": "solution_analyze",
"tool": "solution_analyze",
"action": "verify_empirical_complexity",
"recommended_arguments": {
"problem_dir": str(problem_path),
"solution_type": "sol",
},
"arguments": {
"problem_dir": str(problem_path),
"solution_type": "sol",
},
"priority": "high",
}
)

statement_consistency = self._statement_consistency(problem_path, manifest, tests_manifest)
if statement_consistency["needs_human_review"]:
Expand Down Expand Up @@ -349,6 +378,42 @@ async def _require_special_artifact_gates(
}
)

async def _empirical_complexity_signal(
self, problem_path: Path, manifest: AutoCodeManifest
) -> dict[str, Any]:
sol_source = self._solution_source(problem_path, "sol")
claimed_complexity = None
if sol_source and sol_source.is_file():
code = sol_source.read_text(encoding="utf-8", errors="replace")
claimed_complexity = extract_claimed_complexity(code)

constraints = None
if manifest.constraints:
constraint_numbers = self._constraint_numbers(manifest.constraints)
n_max = max(constraint_numbers) if constraint_numbers else 10000
constraints = {"n_max": n_max, "time_limit_ms": float(manifest.time_limit_ms)}

res = await run_empirical_verification(
str(problem_path),
"sol",
claimed_complexity,
constraints,
)

if res.get("status") == "pending_generator":
return {
"executed": False,
"passed": True,
"evidence": res,
}

passed = bool(res.get("passed"))
return {
"executed": True,
"passed": passed,
"evidence": res,
}

def _statement_consistency(
self, problem_path: Path, manifest: AutoCodeManifest, tests_manifest: dict
) -> dict[str, object]:
Expand Down Expand Up @@ -390,6 +455,12 @@ def _difficulty_signals(
if pattern_complexity:
complexity = self._max_complexity(complexity, pattern_complexity)

empirical_signal = quality_signals.get("empirical_complexity", {})
if self._signal_satisfied(empirical_signal):
fitted_comp = empirical_signal.get("evidence", {}).get("fitted_complexity")
if fitted_comp:
complexity = str(fitted_comp)

constraint_numbers = self._constraint_numbers(manifest.constraints)
n_max = max(constraint_numbers) if constraint_numbers else None
wrong_count = sum(1 for s in manifest.solutions if s.role == "wrong")
Expand Down Expand Up @@ -425,6 +496,8 @@ def _difficulty_signals(
confidence += 0.1
if self._signal_satisfied(quality_signals.get("limit_semantics", {})):
confidence += 0.1
if self._signal_satisfied(empirical_signal):
confidence += 0.1
confidence = min(1.0, confidence)

reasons = [
Expand Down
Loading
Loading