From 25e7c4e41ee360a7d82dba01fc62e3b67aea2be2 Mon Sep 17 00:00:00 2001 From: Saurabh Nandwana Date: Fri, 25 Sep 2026 12:54:48 +0530 Subject: [PATCH] =?UTF-8?q?chore:=20OSS=20standard=20polish=20=E2=80=94=20?= =?UTF-8?q?contributing=20triad,=20repo-root=20hygiene,=20pluto=20shrapnel?= =?UTF-8?q?=20gone,=20canonical=20docs?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Five items, ~1 hour of work, closing every 'below standard' gap surfaced in the pre-v2.10 audit: 1. CONTRIBUTING.md, CODE_OF_CONDUCT.md (Contributor Covenant v2.1), SECURITY.md — the adoptability triad. Email for security reports is dev@motionvector.io with 72h ack. CI expectations written down; the bandit baseline is called a recorded debt list, not something to wave away. 2. Research receipts move from repo root to experiments/routing-research/ (all 8 files: expert_superset_*, prompt_predictor_*, step3b_*, moe_stability_results). They are research receipts behind FLEET-PLAN/CONCEPT claims, not product. README explains what they are and where the live registry actually lives. 3. 'pluto' shrapnel: config.py CORS whitelist had 4 hard-coded pluto.localhost entries (the pre-#101 rename); mflux_driver error text referenced .pluto_config.json which does not exist (real name is .spacepilot_config.json). Both fixed. The LEGACY compat paths (paths.py legacy_user_data_dir/cache_dir, cli.py LEGACY_CONFIG_FILE, PLUTO_MFLUX_BIN env alias) are intentionally kept — they are the 'read-only compat' rule for users whose config predates the rename. tests/test_mflux_driver.py updated to verify the canonical name is honored AND the legacy alias still works when canonical is unset. 4. Canonical-docs headers on the three specs a new contributor hits first: docs/LOCAL-SETUP.md (CANONICAL), docs/BUILD-PLAN.md (STATUS: partially stale, which phases are done as of which date), and docs/design/INFERENCE-SURFACE.md (CANONICAL SPEC for /v1). A reader cold-landing on docs/ now knows which doc is current vs. history without cross-diffing. 5. Regenerated landing/public/registry-snapshot.json from the current main (64 models, 94 variants, 9 measured) after the earlier session's auto-deploy stopcock wiped it. Tests: 1072 passed, 27 skipped locally, full suite. --- CODE_OF_CONDUCT.md | 130 ++++++++++++++++++ CONTRIBUTING.md | 95 +++++++++++++ SECURITY.md | 78 +++++++++++ docs/BUILD-PLAN.md | 7 + docs/LOCAL-SETUP.md | 3 + docs/design/INFERENCE-SURFACE.md | 4 +- experiments/routing-research/README.md | 14 ++ .../expert_superset_results.csv | 0 .../expert_superset_results.json | 0 .../moe_stability_results.json | 0 .../prompt_predictor_results.csv | 0 .../prompt_predictor_results.json | 0 .../step3b_real_flown_results.json | 0 .../step3b_routing_history_results.csv | 0 .../step3b_routing_history_results.json | 0 landing/public/registry-snapshot.json | 2 +- spacepilot/core/config.py | 3 - spacepilot/drivers/mflux_driver.py | 6 +- spacepilot/web/registry.json | 2 +- tests/test_mflux_driver.py | 13 +- 20 files changed, 347 insertions(+), 10 deletions(-) create mode 100644 CODE_OF_CONDUCT.md create mode 100644 CONTRIBUTING.md create mode 100644 SECURITY.md create mode 100644 experiments/routing-research/README.md rename expert_superset_results.csv => experiments/routing-research/expert_superset_results.csv (100%) rename expert_superset_results.json => experiments/routing-research/expert_superset_results.json (100%) rename moe_stability_results.json => experiments/routing-research/moe_stability_results.json (100%) rename prompt_predictor_results.csv => experiments/routing-research/prompt_predictor_results.csv (100%) rename prompt_predictor_results.json => experiments/routing-research/prompt_predictor_results.json (100%) rename step3b_real_flown_results.json => experiments/routing-research/step3b_real_flown_results.json (100%) rename step3b_routing_history_results.csv => experiments/routing-research/step3b_routing_history_results.csv (100%) rename step3b_routing_history_results.json => experiments/routing-research/step3b_routing_history_results.json (100%) diff --git a/CODE_OF_CONDUCT.md b/CODE_OF_CONDUCT.md new file mode 100644 index 0000000..1deeb06 --- /dev/null +++ b/CODE_OF_CONDUCT.md @@ -0,0 +1,130 @@ +# Contributor Covenant Code of Conduct + +## Our Pledge + +We as members, contributors, and leaders pledge to make participation in our +community a harassment-free experience for everyone, regardless of age, body +size, visible or invisible disability, ethnicity, sex characteristics, gender +identity and expression, level of experience, education, socio-economic +status, nationality, personal appearance, race, religion, or sexual identity +and orientation. + +We pledge to act and interact in ways that contribute to an open, welcoming, +diverse, inclusive, and healthy community. + +## Our Standards + +Examples of behavior that contributes to a positive environment for our +community include: + +* Demonstrating empathy and kindness toward other people +* Being respectful of differing opinions, viewpoints, and experiences +* Giving and gracefully accepting constructive feedback +* Accepting responsibility and apologizing to those affected by our + mistakes, and learning from them +* Focusing on what is best not just for us as individuals, but for the + overall community + +Examples of unacceptable behavior include: + +* The use of sexualized language or imagery, and sexual attention or + advances of any kind +* Trolling, insulting or derogatory comments, and personal or political + attacks +* Public or private harassment +* Publishing others' private information, such as a physical or email + address, without their explicit permission +* Other conduct which could reasonably be considered inappropriate in a + professional setting + +## Enforcement Responsibilities + +Community leaders are responsible for clarifying and enforcing our standards +of acceptable behavior and will take appropriate and fair corrective action +in response to any behavior that they deem inappropriate, threatening, +offensive, or harmful. + +Community leaders have the right and responsibility to remove, edit, or +reject comments, commits, code, wiki edits, issues, and other contributions +that are not aligned to this Code of Conduct, and will communicate reasons +for moderation decisions when appropriate. + +## Scope + +This Code of Conduct applies within all community spaces, and also applies +when an individual is officially representing the community in public +spaces. Examples of representing our community include using an official +e-mail address, posting via an official social media account, or acting as +an appointed representative at an online or offline event. + +## Enforcement + +Instances of abusive, harassing, or otherwise unacceptable behavior may be +reported to the community leaders responsible for enforcement at +dev@motionvector.io. All complaints will be reviewed and investigated +promptly and fairly. + +All community leaders are obligated to respect the privacy and security of +the reporter of any incident. + +## Enforcement Guidelines + +Community leaders will follow these Community Impact Guidelines in +determining the consequences for any action they deem in violation of this +Code of Conduct: + +### 1. Correction + +**Community Impact**: Use of inappropriate language or other behavior +deemed unprofessional or unwelcome in the community. + +**Consequence**: A private, written warning from community leaders, +providing clarity around the nature of the violation and an explanation of +why the behavior was inappropriate. A public apology may be requested. + +### 2. Warning + +**Community Impact**: A violation through a single incident or series of +actions. + +**Consequence**: A warning with consequences for continued behavior. No +interaction with the people involved, including unsolicited interaction +with those enforcing the Code of Conduct, for a specified period of time. +This includes avoiding interactions in community spaces as well as external +channels like social media. Violating these terms may lead to a temporary +or permanent ban. + +### 3. Temporary Ban + +**Community Impact**: A serious violation of community standards, +including sustained inappropriate behavior. + +**Consequence**: A temporary ban from any sort of interaction or public +communication with the community for a specified period of time. No public +or private interaction with the people involved, including unsolicited +interaction with those enforcing the Code of Conduct, is allowed during +this period. Violating these terms may lead to a permanent ban. + +### 4. Permanent Ban + +**Community Impact**: Demonstrating a pattern of violation of community +standards, including sustained inappropriate behavior, harassment of an +individual, or aggression toward or disparagement of classes of +individuals. + +**Consequence**: A permanent ban from any sort of public interaction within +the community. + +## Attribution + +This Code of Conduct is adapted from the +[Contributor Covenant](https://www.contributor-covenant.org), version 2.1, +available at +https://www.contributor-covenant.org/version/2/1/code_of_conduct.html. + +Community Impact Guidelines were inspired by +[Mozilla's code of conduct enforcement ladder](https://github.com/mozilla/diversity). + +For answers to common questions about this code of conduct, see the FAQ at +https://www.contributor-covenant.org/faq. Translations are available at +https://www.contributor-covenant.org/translations. diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md new file mode 100644 index 0000000..06cebce --- /dev/null +++ b/CONTRIBUTING.md @@ -0,0 +1,95 @@ +# Contributing to SpacePilot + +Thanks for considering a contribution. The bar is low on ceremony and high on +truth. + +## What this project is + +SpacePilot is a local-first inference orchestrator for single-owner AI +compute fleets. It runs AI models on machines you already own, tells you +honestly what fits before you download it, and records what actually +happened after. It has an OpenAI-compatible `/v1` surface, a FastMCP tool +server for coding agents, a zero-build local web cockpit, and a measured +model registry. + +Two rules govern everything here, from `AGENTS.md`: + +- **Execution over ceremony.** Skip bureaucratic process; bias toward + working code with real receipts (measurements, test output, benchmark + rows). +- **Strict tests before implementation** on production paths. Reproduce + red → fix green → refactor. Exploratory spikes are exempt until they land + in production paths. + +## Project standards + +- **Python 3.11+**, declared in `pyproject.toml` (`requires-python`). +- **Install**: `uv tool install spacepilot` (preferred) or + `pip install spacepilot`. +- **Test locally before a PR:** + + ```bash + python -m pytest tests/ -q + ``` + + CI runs the same suite on GitHub-hosted runners. Do not trust a hardcoded + test count anywhere in the tree — check CI for the current number. + +- **Commit shape**: imperative subject, short body that says *why* not + *what*, refs the PR/issue. No AI-generated filler. `git add` by path, never + `git add -A` — some paths in this repo are build artifacts (`.venv/`, + `build/`, `spacepilot.egg-info/`, `landing/node_modules/`). + +## What a good PR looks like + +- **Scope**: one thing. A PR that changes a driver and a registry row is + fine; a PR that changes the driver and the `/v1` surface and a route table + and adds a feature is three PRs. +- **Tests**: if it changes behavior, it changes tests. A test that passes + against the broken code is not a test — when fixing a bug, prove the new + test *fails* on the original behavior before believing it. +- **Honesty**: every claim in the README, CHANGELOG, docstring, or + `docs/design/*.md` must be true at the time it is written. If a claim + about "works" or "measured" does not survive review, fix the claim, not + the test. + +## Note: `AGENTS.md` is the real rulebook + +`AGENTS.md` at repo root is the instructions this project hands to coding +agents (Claude Code, Codex, Cursor, Copilot, etc.). Everyone works from the +same document — contributors included. Read it before your first PR; it +explains context that is easy to break silently (the pip-less interpreter +rule, the packaging sanity test, the `runtimes` gate, the registry hook that +regenerates `spacepilot/web/registry.json` and the landing snapshot when +model/yaml data moves). + +Run `tools/install_hooks.sh` once after clone to set up `core.hooksPath`. + +## The CI gate and its expectations + +CI runs on every PR and on pushes to `main` / `dev` / `main-*` branches: + +- **pytest** — the full suite +- **packaging** — builds the wheel into a clean venv and runs CLI smoke + tests (defends against undeclared imports and stale build artifacts) +- **security** — `pip-audit` for known CVEs + `bandit` against + `.bandit-baseline.json` (23 known pre-existing findings, recorded; the + gate fails on *new* findings) + +Do not open a PR that skips tests. If your change needs a baseline bump, +say so explicitly in the PR body with the regeneration command and the new +count. + +## Relationship between this repo and its siblings + +| repo | relationship | +|---|---| +| `motionvector-dev/spacepilot` (this one) | the core — CLI, `/v1`, MCP, web cockpit, registry | +| `motionvector-dev/spacebar` | the macOS menu bar app, **read-only** vs SpacePilot — separate cadence, never vendored | +| spacepilot.dev | the landing page, deploys from `landing/` in this repo | + +## Licensing + +Apache-2.0, same license as every registry model card SpacePilot cites. +Contributions on PRs are interpreted identically (Apache-2.0 from the fork +moment onward). diff --git a/SECURITY.md b/SECURITY.md new file mode 100644 index 0000000..b5d116a --- /dev/null +++ b/SECURITY.md @@ -0,0 +1,78 @@ +# Security Policy + +## Reporting a vulnerability + +**Please do not report security vulnerabilities through public GitHub +issues.** + +Report by email to **dev@motionvector.io** with "SECURITY" in the subject +line. Include: + +- the affected version (from `spacepilot --version` or the wheel's + `METADATA`) +- the component (CLI, HTTP `/v1`, FastMCP, dashboard, registry loader) +- reproduction steps or a proof of concept +- what you believe the impact is, if you know + +You will get an acknowledgment within **72 hours**. If a fix requires a +release, we will publish it before any public disclosure, and you will be +credited (or kept anonymous, your choice). + +## What counts as in scope + +- Anything in `spacepilot/` Python code, `spacepilot/web/` static assets, + `native/SpaceBar/` Swift sources, or `landing/` build output. +- The Python packaging path (`pyproject.toml`, `requirements*.txt`) — a + dependency that ships something it should not is a security issue. +- The model registry loader (`spacepilot/model_registry.py`), which parses + YAML from disk — anything that can make it execute or exfiltrate is in + scope. + +## What is NOT in scope + +- **Local-only attacks against the dashboard.** SpacePilot's FastAPI + app is loopback-only by design (`LocalOnlyMiddleware`); attacks that + require the victim to already be on the same machine, or to have + voluntarily installed a malicious local package with access to the + same user's environment, are out of scope for CVE treatment. Those + should go through the normal issue tracker instead. +- **Self-hosted inference boxes** you manually `spacepilot launch` and SSH + into yourself. Those are your own machines under your own credentials. +- **The spot-clip / LTX / Hunyuan DiT engine routes.** They exist as + stubs and every path refuses with 501 — there is no inference path to + attack there in the current release. +- **Social-engineering Saurabh** via email; be verifying, not phishing. + +## If you have already found a remote-code-execution or data-exfiltration +vector + +Do not sit on it. Do not demo it on a live machine. Email us immediately +with the exact steps so we can reproduce; we will not take an aggressive +stance on disclosure timing if you acted in good faith. + +## Known-hardening rules in this repo + +For context, the codebase's own security posture (what is and isn't +trusted): + +- **No Doppler / no secret file committed.** `.env.example` is the shape of + what a deployment wants; real secrets come from `doppler run --` or env + vars only. Anything user-controlled is passed through `X-SpacePilot-Token`. +- **Every compute-spending endpoint requires `X-SpacePilot-Token`** via + `require_token`. Read-only and telemetry routes stay open. + `test_compute_endpoints_all_require_the_token` walks the app and will fail + on any newly-registered mutating route that skips the dependency. +- **`subprocess` calls are argv lists only.** `run_cmd` rejects strings + outright; `shell=True` is never reintroduced. +- **Filesystem paths from a request go through `resolve_output`**, which + resolves and checks containment in `OUTPUTS_DIR`. +- **`innerHTML` on server data goes through `esc()`**; there is a regression + test that fails if server data is interpolated raw. +- **Wheel import gating**: `tests/test_wheel_install.py` asserts no module + in the shipped wheel imports a dependency that is not declared in + `pyproject.toml`. + +## Thanks + +Thanks for spending time on this. A correct security report is worth more +to us than any feature PR. diff --git a/docs/BUILD-PLAN.md b/docs/BUILD-PLAN.md index a1ec9cf..0483544 100644 --- a/docs/BUILD-PLAN.md +++ b/docs/BUILD-PLAN.md @@ -1,5 +1,12 @@ # SpacePilot — Build Plan +> **STATUS: partially stale.** Phases 0.1–0.3 done; the 0.4/0.5 fixes are +> landed as of 2026-09-24 (the honesty pass); Phase 1/3/4/5/6 line numbers +> below were written 2026-08-25 and are approximate — re-grep before +> citing. The doc's own line: 'do not trust a hardcoded count anywhere +> in the tree, check CI for the current number.' + + **Status:** proposed · **Written:** 2026-08-25 · **Verified against:** main `e4b7910` (after PRs #66/#67/#68) · **Audit:** 12 read-only agents, every claim below carries file:line evidence · **Supersedes:** the SpacePilot Build Board artifact diff --git a/docs/LOCAL-SETUP.md b/docs/LOCAL-SETUP.md index 508d1ac..f6a52f3 100644 --- a/docs/LOCAL-SETUP.md +++ b/docs/LOCAL-SETUP.md @@ -1,5 +1,8 @@ # Local setup and upgrades +> **CANONICAL** — this is the authoritative install + upgrade guide as of +> 2026-09-25. Trust this over any blog, README quote, or chat log. + This is the canonical operator guide for installing SpacePilot on a developer machine, selecting inference interpreters, and exposing its MCP server to local agent clients. diff --git a/docs/design/INFERENCE-SURFACE.md b/docs/design/INFERENCE-SURFACE.md index 696610d..f8906bf 100644 --- a/docs/design/INFERENCE-SURFACE.md +++ b/docs/design/INFERENCE-SURFACE.md @@ -1,6 +1,8 @@ # SpacePilot — the inference surface -**Status**: Spec, partly built +> **CANONICAL SPEC** for the HTTP /v1 surface. Status: Spec, partly built; the three /v1 routes and the fit-verdict contract are live (last verified 2026-09-25 against v2.9.0). + +**Written**: 2026-09-02 **Written**: 2026-09-02 **Builds on**: `docs/design/CONCEPT.md` (honesty rules), `docs/design/VISION.md` diff --git a/experiments/routing-research/README.md b/experiments/routing-research/README.md new file mode 100644 index 0000000..1374b3d --- /dev/null +++ b/experiments/routing-research/README.md @@ -0,0 +1,14 @@ +# Routing research receipts + +Files that shipped permanently at the repo root from the pre-flatten +research sessions (2026-08-–09-09). These are **not product code** — they +are the receipts backing specific claims in `docs/design/FLEET-PLAN.md` and +`docs/design/CONCEPT.md` about MoE expert routing, cold-start price +prediction, and step3b routing history. Kept for provenance; not read at +runtime by anything in `spacepilot/`. + +Each file is a self-contained JSON/CSV export from one of the internal +measurement sweeps; see the FLEET-PLAN doc for the sweep that produced +each. If you are looking for the *live* registry, that is +`spacepilot/registry/models/` and the generated +`spacepilot/web/registry.json` — not here. diff --git a/expert_superset_results.csv b/experiments/routing-research/expert_superset_results.csv similarity index 100% rename from expert_superset_results.csv rename to experiments/routing-research/expert_superset_results.csv diff --git a/expert_superset_results.json b/experiments/routing-research/expert_superset_results.json similarity index 100% rename from expert_superset_results.json rename to experiments/routing-research/expert_superset_results.json diff --git a/moe_stability_results.json b/experiments/routing-research/moe_stability_results.json similarity index 100% rename from moe_stability_results.json rename to experiments/routing-research/moe_stability_results.json diff --git a/prompt_predictor_results.csv b/experiments/routing-research/prompt_predictor_results.csv similarity index 100% rename from prompt_predictor_results.csv rename to experiments/routing-research/prompt_predictor_results.csv diff --git a/prompt_predictor_results.json b/experiments/routing-research/prompt_predictor_results.json similarity index 100% rename from prompt_predictor_results.json rename to experiments/routing-research/prompt_predictor_results.json diff --git a/step3b_real_flown_results.json b/experiments/routing-research/step3b_real_flown_results.json similarity index 100% rename from step3b_real_flown_results.json rename to experiments/routing-research/step3b_real_flown_results.json diff --git a/step3b_routing_history_results.csv b/experiments/routing-research/step3b_routing_history_results.csv similarity index 100% rename from step3b_routing_history_results.csv rename to experiments/routing-research/step3b_routing_history_results.csv diff --git a/step3b_routing_history_results.json b/experiments/routing-research/step3b_routing_history_results.json similarity index 100% rename from step3b_routing_history_results.json rename to experiments/routing-research/step3b_routing_history_results.json diff --git a/landing/public/registry-snapshot.json b/landing/public/registry-snapshot.json index 87f34fe..7309e48 100644 --- a/landing/public/registry-snapshot.json +++ b/landing/public/registry-snapshot.json @@ -1,5 +1,5 @@ { - "generated": "2026-09-24", + "generated": "2026-09-25", "rules": { "memory_reserve_fraction": 0.1, "memory_reserve_floor_bytes": 3221225472, diff --git a/spacepilot/core/config.py b/spacepilot/core/config.py index 2d76c60..eb8f318 100644 --- a/spacepilot/core/config.py +++ b/spacepilot/core/config.py @@ -135,11 +135,8 @@ def cors_origins(self) -> list[str]: f"http://localhost:{self.port}", f"http://127.0.0.1:{self.port}", f"http://spacepilot.localhost:{self.port}", - f"http://pluto.localhost:{self.port}", "http://spacepilot.localhost", "https://spacepilot.localhost", - "http://pluto.localhost", - "https://pluto.localhost", ] diff --git a/spacepilot/drivers/mflux_driver.py b/spacepilot/drivers/mflux_driver.py index 1c2257d..618ebdc 100644 --- a/spacepilot/drivers/mflux_driver.py +++ b/spacepilot/drivers/mflux_driver.py @@ -69,7 +69,7 @@ def _default_bin_dir() -> str: def mflux_bin_dir(cfg: Optional[Dict[str, Any]] = None) -> str: """Where the mflux CLI entry points live. Configurable because the path is - this-machine-specific: env var wins, then .pluto_config.json, then the + this-machine-specific: env var wins, then .spacepilot_config.json, then the conda env layout this Mac actually uses.""" from spacepilot.paths import env_value return ( @@ -128,8 +128,8 @@ def _resolve_executable(self, model: str) -> str: if not exe.is_file() or not os.access(exe, os.X_OK): raise MfluxSubprocessError( f"mflux executable not found or not executable: {exe}. " - f"Set SPACEPILOT_MFLUX_BIN (or PLUTO_MFLUX_BIN) or " - f"\"mflux_bin_dir\" in .pluto_config.json to the bin/ directory " + f"Set SPACEPILOT_MFLUX_BIN or " + f"\"mflux_bin_dir\" in .spacepilot_config.json to the bin/ directory " f"of the mflux conda env." ) return str(exe) diff --git a/spacepilot/web/registry.json b/spacepilot/web/registry.json index 87f34fe..7309e48 100644 --- a/spacepilot/web/registry.json +++ b/spacepilot/web/registry.json @@ -1,5 +1,5 @@ { - "generated": "2026-09-24", + "generated": "2026-09-25", "rules": { "memory_reserve_fraction": 0.1, "memory_reserve_floor_bytes": 3221225472, diff --git a/tests/test_mflux_driver.py b/tests/test_mflux_driver.py index 7550c4a..ac19992 100644 --- a/tests/test_mflux_driver.py +++ b/tests/test_mflux_driver.py @@ -83,11 +83,22 @@ def test_command_for_alias_routes_flux2_klein_to_its_own_entry_point(): def test_bin_dir_prefers_env_var(monkeypatch): - monkeypatch.setenv("PLUTO_MFLUX_BIN", "/opt/custom-mflux/bin") + monkeypatch.setenv("SPACEPILOT_MFLUX_BIN", "/opt/custom-mflux/bin") assert mflux_bin_dir() == "/opt/custom-mflux/bin" +def test_bin_dir_prefers_the_legacy_pluto_env_name_too(monkeypatch): + """PLUTO_MFLUX_BIN was the pre-rename env name; it is kept as a silent + alias so an existing setup does not break on upgrade. The canonical name + must be REMOVED, not empty — env_value reads the empty string as a real + value and stops there.""" + monkeypatch.delenv("SPACEPILOT_MFLUX_BIN", raising=False) + monkeypatch.setenv("PLUTO_MFLUX_BIN", "/opt/pluto-mflux/bin") + assert mflux_bin_dir() == "/opt/pluto-mflux/bin" + + def test_bin_dir_falls_back_to_config(monkeypatch): + monkeypatch.delenv("SPACEPILOT_MFLUX_BIN", raising=False) monkeypatch.delenv("PLUTO_MFLUX_BIN", raising=False) assert mflux_bin_dir({"mflux_bin_dir": "/opt/other/bin"}) == "/opt/other/bin"