From 34fe4bac3ae26f7a900699a4ea081f3f5f4006d0 Mon Sep 17 00:00:00 2001 From: AlbertXXuu Date: Mon, 31 Aug 2026 17:08:12 +0800 Subject: [PATCH] fix: prepare v1.1.2 maintenance patch --- CHANGELOG.md | 23 ++++++++ README.md | 10 ++-- README.zh-CN.md | 8 +-- TASKS.md | 6 +- docs/08-risks-and-decisions.md | 6 +- docs/MAINTENANCE.md | 2 +- docs/evaluation-protocol.md | 7 +++ docs/post-v1-roadmap.md | 9 ++- docs/reports/v1.1.2-patch-validation.md | 57 +++++++++++++++++++ pyproject.toml | 2 +- src/openmultimodal_lab/__init__.py | 2 +- .../assets/studio/alvenx-studio.css | 3 +- src/openmultimodal_lab/studio_assets.py | 2 +- tests/test_package_metadata.py | 42 +++++++++++++- tests/test_studio.py | 8 ++- 15 files changed, 164 insertions(+), 23 deletions(-) create mode 100644 docs/reports/v1.1.2-patch-validation.md diff --git a/CHANGELOG.md b/CHANGELOG.md index 672ca35..c94db00 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,29 @@ All notable changes to this project will be documented in this file. +## [1.1.2] - 2026-08-31 + +### Fixed + +- Remove framework-added padding and button margins from the Studio header navigation so the + packaged interface retains the canonical AlvenX header height across the audited desktop widths. + +### Clarified + +- Define the current CUDA `peak_gpu_memory_mb` value as PyTorch allocator memory rather than total + process or device VRAM, and require alternative measurement boundaries to remain separate. +- Require an observed low-memory or latency problem before adding a quantized or alternative + runtime comparison, with matched revisions, tasks, prompts, generation settings, and metrics. +- Record that useful boundaries from two early documentation-only project remnants were retained + without merging their abandoned scope into OpenMultimodalLab. + +The immutable `v1.0.0` research tag, datasets, raw results, report bundle, and published +measurements remain unchanged. Source validation is recorded in +`docs/reports/v1.1.2-patch-validation.md`; final remote provenance belongs to the AlvenX +post-closure audit and GitHub Release. + +Compare: [`v1.1.1...v1.1.2`](https://github.com/AlbertXXuu/OpenMultimodalLab/compare/v1.1.1...v1.1.2) + ## [1.1.1] - 2026-08-31 ### Fixed diff --git a/README.md b/README.md index 86466d6..062d3e9 100644 --- a/README.md +++ b/README.md @@ -12,10 +12,10 @@ [简体中文](README.zh-CN.md) -> Version status: current software is the `v1.1.1` closure maintenance release. The research and -> evidence baseline remains the immutable `v1.0.0` public release. The closure line adds -> presentation and maintenance work, not a new benchmark claim; `v1.1.1` corrects the documented -> current-software clone target. See the +> Version status: current software is the `v1.1.2` maintenance patch. The research and evidence +> baseline remains the immutable `v1.0.0` public release. This patch aligns the packaged Studio +> header with the canonical cross-product geometry and clarifies measurement and future-runtime +> boundaries; it does not add benchmark evidence or change published results. See the > [maintenance policy](docs/MAINTENANCE.md) and [portfolio evidence](PORTFOLIO.md). A local-first, reproducible benchmark toolkit for answering a practical @@ -153,7 +153,7 @@ The core path does not download a model and works on Python 3.11, 3.12, or 3.13. The primary examples use Windows PowerShell: ```powershell -git clone --branch v1.1.1 --depth 1 https://github.com/AlbertXXuu/OpenMultimodalLab.git +git clone --branch v1.1.2 --depth 1 https://github.com/AlbertXXuu/OpenMultimodalLab.git cd OpenMultimodalLab py -3.11 -m venv .venv diff --git a/README.zh-CN.md b/README.zh-CN.md index 9148c59..5495e4d 100644 --- a/README.zh-CN.md +++ b/README.zh-CN.md @@ -12,9 +12,9 @@ [English](README.md) -> 版本状态:当前软件是 `v1.1.1` 收尾维护版;研究与证据基线仍是不可变的 `v1.0.0` -> 公开正式版。收尾版本只处理展示与维护,不产生新的基准结论;`v1.1.1` 修正文档中的 -> 当前软件克隆目标。详见 +> 版本状态:当前软件是 `v1.1.2` 维护修补版。研究与证据基线仍是不可变的 `v1.0.0` +> 公开正式版。本次修补让打包后的 Studio Header 回到跨产品统一几何,并澄清测量与 +> 未来运行时边界;它不新增基准证据,也不改变已发布结果。详见 > [维护政策](docs/MAINTENANCE.md)与[作品证据页](PORTFOLIO.md)。 一个本地优先、强调可复现证据的多模态模型评测工具。它要回答的不是“哪个 @@ -132,7 +132,7 @@ flowchart LR 核心路径不会下载模型,可在 Python 3.11、3.12 或 3.13 运行: ```powershell -git clone --branch v1.1.1 --depth 1 https://github.com/AlbertXXuu/OpenMultimodalLab.git +git clone --branch v1.1.2 --depth 1 https://github.com/AlbertXXuu/OpenMultimodalLab.git cd OpenMultimodalLab py -3.11 -m venv .venv diff --git a/TASKS.md b/TASKS.md index 070d636..63736ec 100644 --- a/TASKS.md +++ b/TASKS.md @@ -1,8 +1,8 @@ # 当前任务清单 -> `v1.0.0` 研究与证据基线已于 2026-08-10 正式发布,`v1.1.1` 展示与维护收尾版 -> 已于 2026-08-31 发布。后续工作不改写 v1.0.0 的标签、原始结果或证据;当前进入 -> 维护模式,不产生新的研究主张。 +> `v1.0.0` 研究与证据基线已于 2026-08-10 正式发布;当前公开维护软件是 `v1.1.2`。 +> 该展示与维护修补版不改写 v1.0.0 的标签、原始结果或证据;项目保持维护模式, +> 不产生新的研究主张。 ## Now:发布后可靠性审查 diff --git a/docs/08-risks-and-decisions.md b/docs/08-risks-and-decisions.md index d6fbaee..2af258d 100644 --- a/docs/08-risks-and-decisions.md +++ b/docs/08-risks-and-decisions.md @@ -11,7 +11,8 @@ - v1.0 正式实验已在该 8GB GPU 上完成;可用磁盘空间属于运行时状态,不在 文档中长期固定。 - 维护目标是每周 2~3 次有证据的更新;具体投入时间按学业安排调整。 -- 现有工作区内已有 LocalModelLab 和 AgentReliabilityLab,本项目不覆盖它们。 +- 早期 LocalModelLab 与 AgentReliabilityLab 文档中的有效方法边界已分别提炼到本项目路线图、 + 评测协议和 AlvenX incubator;旧的无代码研究残余已清理,不再构成并行项目。 ## 2. 当前假设 @@ -44,6 +45,9 @@ 原因:三者研究问题不同,独立仓库更容易形成清楚的 README、路线图和面试故事。 +这是创建仓库时的历史决定。后续整理仅提炼仍有效的方法约束并清理未实现旧文档,未把旧路线图或 +通用 Agent 平台范围并入 OpenMultimodalLab。 + ### D-002:先做评测核心 第一阶段使用确定性 `mock` 后端,不立即安装大型模型。 diff --git a/docs/MAINTENANCE.md b/docs/MAINTENANCE.md index 8d099ac..192b255 100644 --- a/docs/MAINTENANCE.md +++ b/docs/MAINTENANCE.md @@ -1,6 +1,6 @@ # Maintenance policy -Current public release: `v1.1.1` +Current public and maintained release: `v1.1.2` Research/evidence baseline: immutable `v1.0.0` Development mode: maintenance; no new research claim diff --git a/docs/evaluation-protocol.md b/docs/evaluation-protocol.md index 6373b0b..13f5cbd 100644 --- a/docs/evaluation-protocol.md +++ b/docs/evaluation-protocol.md @@ -233,6 +233,13 @@ The native Transformers visual-text adapters record the same fields: - `latency_ms`: adapter invocation plus deterministic evaluation in the runner; - `peak_gpu_memory_mb`: maximum CUDA memory allocated during generation. +For the current PyTorch CUDA adapters, `peak_gpu_memory_mb` means allocator +memory, not total process VRAM or the peak for the whole device. An alternative +runtime must declare its measurement boundary. An NVML measurement must retain +the idle baseline, sampling interval, and whether another process contaminated +the observation. Results from different memory boundaries must not share one +memory ranking. + `output_tokens_per_second` counts generated token IDs, including terminal special tokens, over `generation_ms`. `decode_tokens_per_second` excludes the first generated token and divides by `generation_ms - ttft_ms`. diff --git a/docs/post-v1-roadmap.md b/docs/post-v1-roadmap.md index a1d8bcb..3948dcf 100644 --- a/docs/post-v1-roadmap.md +++ b/docs/post-v1-roadmap.md @@ -61,8 +61,13 @@ An adapter that only makes another model import successfully is not enough. ### Later: conditional scope -- Quantized or alternative runtimes when they enable a measured hardware or - latency use case. +- Quantized or alternative runtimes only after a concrete low-memory or latency + problem is observed. Compare quality, TTFT, decode speed, memory, OOM behavior, + and effective context under the same immutable model revision, task set, + prompt/chat template, and generation protocol. Differences across GGUF, AWQ, + GPTQ, NF4, or different backends cannot be attributed to bit width alone. + Long-context/OOM gradients and 50-run stability checks begin only after a + candidate signal exists; they are not a standing benchmark matrix. - Additional task families when their media can be regenerated and reviewed. - A UI or remote execution layer only when CLI user evidence shows it removes a real barrier without weakening provenance. diff --git a/docs/reports/v1.1.2-patch-validation.md b/docs/reports/v1.1.2-patch-validation.md new file mode 100644 index 0000000..7e1ce51 --- /dev/null +++ b/docs/reports/v1.1.2-patch-validation.md @@ -0,0 +1,57 @@ +# OpenMultimodalLab v1.1.2 patch validation + +- Status: **VALIDATED RELEASE SOURCE** +- Source validated: `2026-08-31` (`Asia/Shanghai`) +- Owner authorization: AlvenX workspace decision `D-032`, recorded `2026-08-31` +- Release identity: `v1.1.2` +- Research/evidence baseline: immutable `v1.0.0` + +## Patch meaning + +This maintenance patch addresses one observed Studio layout defect and three bounded documentation +clarifications: + +1. The Studio header navigation no longer retains framework-added outer padding or button margins, + so its shell returns to the canonical AlvenX product-header height. +2. The current CUDA `peak_gpu_memory_mb` field is explicitly scoped to PyTorch allocator memory, + not total process VRAM or whole-device peak usage. Different measurement boundaries remain + separate rather than being mixed into one memory ranking. +3. Quantized or alternative runtimes enter the roadmap only after an observed memory or latency + problem and must use matched revisions, tasks, prompts, generation settings, and reported + boundaries. +4. The project decision record now states that useful constraints from two early documentation-only + remnants were retained without absorbing their abandoned product scope. + +The patch does not add a model, adapter, dependency, task, dataset, schema, benchmark run, or +research conclusion. + +## Frozen v1.0.0 surface + +The annotated `v1.0.0` tag, reviewed task and media identities, pinned model revisions, formal run +configuration, raw JSONL results, manifests, report bundle, and published measurements are not +modified. The evaluation-protocol edit is a post-release measurement-boundary clarification; it +does not rewrite the protocol or evidence bytes reachable from `v1.0.0`, and it does not recompute +or reinterpret the released numeric results. + +`CITATION.cff`, `docs/release-approvals.json`, historical release audits, dependencies, and GitHub +Actions workflows remain unchanged. + +## Source verification + +| Gate | Result | Evidence | +| --- | --- | --- | +| Version identity | PASS | Package metadata, import version, Studio badge, bilingual README, changelog, maintenance policy, task status, and this audit identify software `v1.1.2`; evidence stays `v1.0.0`. | +| Targeted package and Studio tests | PASS | Pytest collected `29` cases on Python `3.13.5`: `27` passed and `2` optional-environment cases skipped. The Header padding/margin regression and cross-surface version consistency test passed. | +| Complete local unit suite | PASS | `unittest discover` executed `202` cases on Python `3.13.5` / Windows 11 build `26200`: `199` passed and `3` optional-environment cases skipped. A separate pytest collection covered `207` cases: `204` passed and `3` skipped. | +| Repository and evidence checks | PASS | Repository audit: `195` text files, `278` Markdown links, and `1,056` JSON/JSONL documents. Report verification matched `4` sources and `5` outputs; runtime-license policy passed; offline contributor smoke completed `3/3` records and reported package `1.1.2`. | +| Frozen v1.0 technical gate | PASS, bounded scope | The existing technical-strict readiness check passed `18/18`; it revalidates the immutable v1.0 evidence contract without changing the released evidence. | +| AlvenX brand/workspace validation | PASS | Brand revision `2026-08-24.1` validated `14` hashed assets, `2` canonical masters, and `3` wordmark lockups; workspace revision `2026-08-30.1` passed. | +| Interface regression review | PASS | Edge 152, Chrome 151, and Firefox 153 passed 900/1024/1280/1440/1600 px with the canonical 74.48 px Header, 44 px controls, and no horizontal overflow. | + +## Remote provenance boundary + +This in-repository record validates the release source without embedding a commit identity that can +only be known after the source is committed. The final remote commit, GitHub Actions runs, +distribution filenames and hashes, and installed-artifact checks are recorded in the workspace-level +AlvenX post-closure audit and the GitHub Release. Those external records bind the published source +and artifacts; this document does not manufacture that provenance in advance. diff --git a/pyproject.toml b/pyproject.toml index c20053b..2476bcc 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "openmultimodal-lab" -version = "1.1.1" +version = "1.1.2" description = "The local-first, reproducible multimodal benchmark engine behind AlvenX." readme = "README.md" requires-python = ">=3.11" diff --git a/src/openmultimodal_lab/__init__.py b/src/openmultimodal_lab/__init__.py index 52b089e..55b2d19 100644 --- a/src/openmultimodal_lab/__init__.py +++ b/src/openmultimodal_lab/__init__.py @@ -3,4 +3,4 @@ from .models import EvaluationTask, ModelOutput, RunRecord, ScoringConfig __all__ = ["EvaluationTask", "ModelOutput", "RunRecord", "ScoringConfig"] -__version__ = "1.1.1" +__version__ = "1.1.2" diff --git a/src/openmultimodal_lab/assets/studio/alvenx-studio.css b/src/openmultimodal_lab/assets/studio/alvenx-studio.css index 4f86b2f..def3395 100644 --- a/src/openmultimodal_lab/assets/studio/alvenx-studio.css +++ b/src/openmultimodal_lab/assets/studio/alvenx-studio.css @@ -122,13 +122,14 @@ body { margin: 0; } display: flex; gap: 5px; margin-left: auto; - padding: 4px; + padding: 0; border-radius: 999px; background: rgb(255 255 255 / 28%); } .header-nav button { min-width: 44px; min-height: 44px; + margin: 0 !important; padding: 9px 13px; border: 0; border-radius: 999px; diff --git a/src/openmultimodal_lab/studio_assets.py b/src/openmultimodal_lab/studio_assets.py index c133dc5..05dcbda 100644 --- a/src/openmultimodal_lab/studio_assets.py +++ b/src/openmultimodal_lab/studio_assets.py @@ -103,7 +103,7 @@ def _studio_theme() -> str: - Studio v1.1.1 · Evidence v1.0.0 + Studio v1.1.2 · Evidence v1.0.0 """.strip() diff --git a/tests/test_package_metadata.py b/tests/test_package_metadata.py index 9e5ecd0..8af96e1 100644 --- a/tests/test_package_metadata.py +++ b/tests/test_package_metadata.py @@ -5,24 +5,62 @@ from pathlib import Path from openmultimodal_lab import __version__ +from openmultimodal_lab.studio_assets import BRAND_HEADER_HTML PROJECT_ROOT = Path(__file__).resolve().parents[1] +CURRENT_SOFTWARE_VERSION = "1.1.2" class PackageMetadataTests(unittest.TestCase): - def test_candidate_version_matches_closure_version(self) -> None: + def test_current_version_matches_maintenance_patch(self) -> None: pyproject = tomllib.loads( (PROJECT_ROOT / "pyproject.toml").read_text(encoding="utf-8") ) - self.assertEqual(__version__, "1.1.1") + self.assertEqual(__version__, CURRENT_SOFTWARE_VERSION) self.assertEqual(pyproject["project"]["version"], __version__) self.assertIn( "Development Status :: 4 - Beta", pyproject["project"]["classifiers"], ) + def test_current_software_version_surfaces_agree(self) -> None: + tagged_version = f"v{CURRENT_SOFTWARE_VERSION}" + expected_markers = { + "README.md": ( + f"current software is the `{tagged_version}` maintenance patch", + f"git clone --branch {tagged_version} --depth 1", + ), + "README.zh-CN.md": ( + f"当前软件是 `{tagged_version}` 维护修补版", + f"git clone --branch {tagged_version} --depth 1", + ), + "CHANGELOG.md": ( + f"## [{CURRENT_SOFTWARE_VERSION}] - 2026-08-31", + f"compare/v1.1.1...{tagged_version}", + ), + "docs/MAINTENANCE.md": ( + f"Current public and maintained release: `{tagged_version}`", + ), + "TASKS.md": (f"当前公开维护软件是 `{tagged_version}`",), + "docs/reports/v1.1.2-patch-validation.md": ( + f"# OpenMultimodalLab {tagged_version} patch validation", + "Status: **VALIDATED RELEASE SOURCE**", + ), + } + + for relative_path, markers in expected_markers.items(): + text = (PROJECT_ROOT / relative_path).read_text(encoding="utf-8") + for marker in markers: + with self.subTest(path=relative_path, marker=marker): + self.assertIn(marker, text) + + self.assertIn( + f"Studio {tagged_version} · Evidence v1.0.0", + BRAND_HEADER_HTML, + ) + def test_public_attribution_uses_the_owner_approved_identity(self) -> None: pyproject = tomllib.loads( (PROJECT_ROOT / "pyproject.toml").read_text(encoding="utf-8") diff --git a/tests/test_studio.py b/tests/test_studio.py index 487d6ec..1c46ffc 100644 --- a/tests/test_studio.py +++ b/tests/test_studio.py @@ -324,11 +324,17 @@ def test_brand_assets_are_local_and_wordmark_is_not_interactive(self) -> None: self.assertIn('data-studio-tab="run"', BRAND_HEADER_HTML) self.assertIn('data-studio-tab="reports"', BRAND_HEADER_HTML) self.assertIn('data-studio-tab="method"', BRAND_HEADER_HTML) - self.assertIn("Studio v1.1.1 · Evidence v1.0.0", BRAND_HEADER_HTML) + self.assertIn("Studio v1.1.2 · Evidence v1.0.0", BRAND_HEADER_HTML) self.assertIn("Evidence v1.0.0", BRAND_HEADER_HTML) self.assertIn('document.title = "OpenMultimodalLab · AlvenX"', STUDIO_NAV_JS) self.assertIn("top: 14px", STUDIO_CSS) self.assertGreaterEqual(STUDIO_CSS.count("z-index: 100"), 2) + header_nav_css = STUDIO_CSS.split(".header-nav {", 1)[1].split("}", 1)[0] + header_nav_button_css = STUDIO_CSS.split(".header-nav button {", 1)[1].split( + "}", 1 + )[0] + self.assertIn("padding: 0;", header_nav_css) + self.assertIn("margin: 0 !important;", header_nav_button_css) self.assertIn( "width: calc(min(100%, 1480px) - 2 * clamp(18px, 5vw, 74px))", STUDIO_CSS,