Files
scrivas/scripts/ec2_code_inspect.py
T
Alvaro Del Valle 44d4bf6748 Add EC2 source-code discovery and reliability triage
Second discovery pass over the Scrivas EC2 estate (716468089330, us-east-2,
7 instances), prompted by the client reporting reliability issues and by their
lack of access to source code held under contract by the incumbent vendor.

Source code recovery
- All 10 application repositories exist as complete git checkouts on
  Scrivas-owned instances, with full history rather than deployed artifacts:
  4 app repos on Scrivas_dev_env, 6 ML repos on ML_dev.
- Every remote points at git@git.devteam.space (the contractor's self-hosted
  GitLab), which Scrivas does not control. The on-instance checkouts are the
  client's only independent leverage over their own source.
- Gap: both /var/www frontends are build output with no .git, so frontend
  source is not recoverable from EC2.
- Time-sensitive: scrivas_backend received a commit on the assessment date.

Reliability triage
- Production runs 23 containers on a single 15 GiB host, including 3 Postgres
  instances, Kafka and OpenSearch, at 73% memory at rest with no per-container
  memory limits and no swap on any of the 7 instances.
- Kafka, OpenSearch and search-api carry restart policy `no`, so a host reboot
  yields a partially-recovered stack.
- Recorded as a structural exposure, not an observed root cause: no OOM event
  is present in retained logs and RestartCount is 0 on every prod container.
  Confirming the hypothesis needs CloudWatch history the boxes do not retain.

Contents
- scripts/ec2_code_discovery.py  EC2 inventory (describe + user data)
- scripts/ec2_code_inspect.py    read-only SSM probe set, reviewable in PROBES
- findings/ec2_code_discovery_report.md   narrative writeup
- findings/code_dashboard.html            client-facing dashboard
- findings/ec2_code_inspect*.json         raw probe output
- index.html                              links the new dashboard and evidence

All access was read-only: no writes, restarts or config changes on any
instance. Probe output was scanned for credentials before commit; git metadata
was read as the owning user rather than by writing a safe.directory entry.

Claude-Session: https://claude.ai/code/session_01YMxVaHXJsqpqKwncNQ9b1e
2026-08-28 16:24:24 -04:00

159 lines
6.5 KiB
Python

#!/usr/bin/env python3
"""
Scrivas — EC2 code & reliability inspection (stage 2)
=====================================================
Runs a fixed, read-only command set on SSM-managed instances to locate the
deployed application code and collect reliability evidence.
Context: Scrivas does not hold the source for their own platform (built by a
third-party contractor) and is reporting production reliability issues. This
inspects the client's own instances, in the client's own account, to (a) find
where the code lives and (b) gather triage evidence.
SCOPE NOTE: ssm:SendCommand executes on the host. The command set below is
read-only by construction -- no writes, no restarts, no config changes, and
no dumping of file *contents* beyond manifests and VCS metadata. Review
PROBES before running. Requires explicit operator approval.
Usage:
python3 scripts/ec2_code_inspect.py --list
python3 scripts/ec2_code_inspect.py --instance i-073154fb4fa773bbd
python3 scripts/ec2_code_inspect.py --all
Output: findings/ec2_code_inspect.json
"""
import argparse
import json
import os
import time
from datetime import datetime, date
import boto3
from botocore.exceptions import ClientError, BotoCoreError
DEFAULT_PROFILE = "dasnuve-scrivas-louis-impersonation"
REGION = "us-east-2"
FINDINGS = os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "findings")
# Read-only probes. Each is (label, shell). Keep every command non-mutating.
PROBES = [
("os_release", "cat /etc/os-release; uname -a"),
("uptime_load", "uptime; cat /proc/loadavg"),
("disk", "df -h; echo '--- inodes'; df -i"),
("memory", "free -h; echo '--- swap'; swapon --show"),
("code_dirs", "ls -la /opt /srv /var/www /home 2>/dev/null"),
("git_checkouts", "find / -maxdepth 6 -name .git -type d "
"-not -path '*/node_modules/*' 2>/dev/null | head -40"),
("git_remotes", "for g in $(find / -maxdepth 6 -name .git -type d "
"-not -path '*/node_modules/*' 2>/dev/null | head -20); do "
"r=$(dirname $g); echo \"== $r\"; "
"git -C $r remote -v 2>/dev/null; "
"git -C $r log -1 --format='%H %ad %an %s' 2>/dev/null; "
"git -C $r status -sb 2>/dev/null | head -5; done"),
("manifests", "find / -maxdepth 6 \\( -name package.json -o -name requirements.txt "
"-o -name pyproject.toml -o -name go.mod -o -name Dockerfile "
"-o -name docker-compose.y*ml \\) -not -path '*/node_modules/*' "
"2>/dev/null | head -40"),
("processes", "ps auxww --sort=-%mem | head -30"),
("listening", "ss -tulpnH 2>/dev/null | head -40"),
("systemd_units", "systemctl list-units --type=service --state=running --no-pager --no-legend | head -40"),
("systemd_failed", "systemctl list-units --state=failed --no-pager --no-legend"),
("docker", "docker ps -a --format '{{.Names}}\t{{.Image}}\t{{.Status}}' 2>/dev/null | head -30"),
("docker_images", "docker images --format '{{.Repository}}:{{.Tag}}\t{{.CreatedAt}}' 2>/dev/null | head -20"),
("oom_kills", "sudo dmesg -T 2>/dev/null | grep -iE 'oom|killed process' | tail -20"),
("svc_restarts", "sudo journalctl --since '7 days ago' --no-pager 2>/dev/null "
"| grep -iE 'segfault|out of memory|failed with result|start-limit' | tail -40"),
("reboots", "last -x reboot 2>/dev/null | head -10"),
("cron", "ls -la /etc/cron.d 2>/dev/null; crontab -l 2>/dev/null"),
]
def _default(o):
return o.isoformat() if isinstance(o, (datetime, date)) else str(o)
def managed(session):
ssm = session.client("ssm", region_name=REGION)
out = []
for page in ssm.get_paginator("describe_instance_information").paginate():
out += [i for i in page.get("InstanceInformationList", [])
if i.get("PingStatus") == "Online"]
return out
def run_probe(ssm, iid, label, shell, timeout=120):
try:
cmd = ssm.send_command(
InstanceIds=[iid],
DocumentName="AWS-RunShellScript",
Comment=f"dasnuve-discovery:{label}"[:100],
Parameters={"commands": [shell], "executionTimeout": [str(timeout)]},
)["Command"]["CommandId"]
except (ClientError, BotoCoreError) as e:
return {"error": str(e)}
for _ in range(int(timeout / 2)):
time.sleep(2)
try:
r = ssm.get_command_invocation(CommandId=cmd, InstanceId=iid)
except ClientError as e:
if "InvocationDoesNotExist" in str(e):
continue
return {"error": str(e)}
if r["Status"] in ("Pending", "InProgress", "Delayed"):
continue
return {
"status": r["Status"],
"stdout": r.get("StandardOutputContent", "").rstrip(),
"stderr": r.get("StandardErrorContent", "").rstrip(),
}
return {"error": "timed out waiting for invocation"}
def inspect(session, iid):
ssm = session.client("ssm", region_name=REGION)
print(f"\n=== {iid}")
res = {}
for label, shell in PROBES:
print(f" .. {label}", flush=True)
res[label] = run_probe(ssm, iid, label, shell)
return res
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--profile", default=DEFAULT_PROFILE)
ap.add_argument("--instance", action="append", dest="instances")
ap.add_argument("--all", action="store_true")
ap.add_argument("--list", action="store_true")
args = ap.parse_args()
session = boto3.Session(profile_name=args.profile)
online = managed(session)
if args.list:
for i in online:
print(f'{i["InstanceId"]:22} {i["PingStatus"]:8} '
f'{i.get("PlatformName")} {i.get("PlatformVersion")}')
return
targets = args.instances or ([i["InstanceId"] for i in online] if args.all else [])
if not targets:
ap.error("pass --instance ID (repeatable), --all, or --list")
report = {
"generated": datetime.now().astimezone().isoformat(),
"region": REGION,
"probes": [p[0] for p in PROBES],
"results": {iid: inspect(session, iid) for iid in targets},
}
os.makedirs(FINDINGS, exist_ok=True)
path = os.path.join(FINDINGS, "ec2_code_inspect.json")
with open(path, "w") as fh:
json.dump(report, fh, indent=2, default=_default)
print(f"\nwrote {path}")
if __name__ == "__main__":
main()