Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -87,5 +87,9 @@ functions-python/**/*.csv
# Project files
*.code-workspace

# Local build output, including the default destination of
# scripts/parquet-generate-local.sh
.dist/

# Ignore OpenApi local backup files
*.yaml.bak
14 changes: 14 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -181,6 +181,20 @@ Note: the tests rely on having an empty local test DB instance. If you have data
./scripts/docker-localdb-rebuild-data.sh --use-test-db
```

### Browsing a feed locally (Parquet)

To convert a GTFS feed to Parquet on this machine - no bucket, database or GCP
credentials - and optionally serve it to a browser:

```bash
scripts/parquet-generate-local.sh mdb-1210 --serve
```

The script accepts a feed or dataset stable id, a `.zip`, or an unpacked folder. See
[functions-python/parquet_builder](functions-python/parquet_builder/README.md) for the
full local testing guide, including running it against the Operations API and the
operations web app.


## Running with Docker

Expand Down
38 changes: 36 additions & 2 deletions api/src/shared/common/gcp_memory_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,10 +3,31 @@
import resource
import shutil
import sys
from dataclasses import dataclass
from typing import Optional

MB_MULTIPLIER = 1024**2


@dataclass(frozen=True)
class MemoryBudget:
"""What this container was given, as opposed to what it went on to use.

These three numbers are computed on the way to setting RLIMIT_AS and were previously
only logged. A process that records how much memory a job needed cannot say whether
that was comfortable without also recording what it had, and reconstructing the
budget afterwards from the deployment means trusting that the deployment has not
moved since.

Any field may be None where the figure could not be read, which is normal off
Cloud Run.
"""

cgroup_limit_bytes: Optional[int] = None
volume_bytes: Optional[int] = None
rlimit_as_bytes: Optional[int] = None


def is_filesystem_tmpfs(mount_point):
"""
Check if the given mount_point is a tmpfs filesystem in /proc/mounts.
Expand Down Expand Up @@ -139,12 +160,20 @@ def limit_gcp_memory(mount_point):

Environment Variables:
MEMORY_MARGIN_MB: Safety margin in megabytes (default: 200)

Returns:
A MemoryBudget describing what was found and what was set. Fields are None where
a figure could not be read or no limit was applied; callers that only want the
side effect can ignore it.
"""
cgroup_limit_bytes = get_memory_limit_cgroup_bytes()
volume_bytes = get_tmpfs_size_bytes(mount_point)

# Calculate available memory: cgroup limit - tmpfs size
available_memory_bytes = get_available_process_memory_bytes(mount_point)
if not available_memory_bytes or available_memory_bytes <= 0:
logging.info("Could not find the total memory of the process. Memory limit not set.")
return
return MemoryBudget(cgroup_limit_bytes=cgroup_limit_bytes, volume_bytes=volume_bytes)

# Parse and validate the memory margin
memory_margin_mb = 200
Expand Down Expand Up @@ -177,9 +206,14 @@ def limit_gcp_memory(mount_point):
"Computed RLIMIT_AS <= 0 (%.2f MiB). Skipping setrlimit.",
mem_limit / MB_MULTIPLIER,
)
return
return MemoryBudget(cgroup_limit_bytes=cgroup_limit_bytes, volume_bytes=volume_bytes)

# Set RLIMIT_AS (address space limit) to prevent OOM kills
# When this limit is exceeded, Python will raise MemoryError instead of being killed
resource.setrlimit(resource.RLIMIT_AS, (mem_limit, mem_limit))
logging.info("RLIMIT_AS set to %.2f MiB", mem_limit / MB_MULTIPLIER)
return MemoryBudget(
cgroup_limit_bytes=cgroup_limit_bytes,
volume_bytes=volume_bytes,
rlimit_as_bytes=mem_limit,
)
9 changes: 8 additions & 1 deletion api/src/shared/common/gcp_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -258,8 +258,13 @@ def create_http_task_with_name(
task_time,
http_method: any, # tasks_v2.HttpMethod
timeout_s: int = 1800, # 30 minutes
raise_on_error: bool = False,
):
"""Creates a GCP Cloud Task."""
"""Creates a GCP Cloud Task.

`raise_on_error` is for callers that record state on the strength of the enqueue.
Off by default: the original callers are fire-and-forget.
"""
from google.cloud import tasks_v2
from google.protobuf import duration_pb2

Expand Down Expand Up @@ -290,3 +295,5 @@ def create_http_task_with_name(
logging.info("Task already exists for %s, skipping.", task_name)
else:
logging.error("Error creating task: %s", e)
if raise_on_error:
raise
Loading
Loading