Database Concurrency Optimizations (#80)

* fix: increase db pool limit and optimize crud requests

* fix: Sentry tracing and fly concurrency

* feat: Add alembic and indexes
This commit is contained in:
Vineeth Voruganti 2024-12-13 11:56:39 -05:00 committed by GitHub
parent dcff9e31f6
commit 57ef19315e
No known key found for this signature in database
GPG Key ID: B5690EEEBB952194
15 changed files with 538 additions and 177 deletions

View File

@ -7,6 +7,11 @@ and this project adheres to [Semantic Versioning](http://semver.org/).
## [0.0.15]
### Added
- Alembic for handling database migrations
- Additional indexes for reading Messages and Metamessages
### Fixed
- Dialectic Streaming Endpoint properly sends text in `StreamingResponse`

117
alembic.ini Normal file
View File

@ -0,0 +1,117 @@
# A generic, single database configuration.
[alembic]
# path to migration scripts
# Use forward slashes (/) also on windows to provide an os agnostic path
script_location = migrations
# template used to generate migration file names; The default value is %%(rev)s_%%(slug)s
# Uncomment the line below if you want the files to be prepended with date and time
# see https://alembic.sqlalchemy.org/en/latest/tutorial.html#editing-the-ini-file
# for all available tokens
# file_template = %%(year)d_%%(month).2d_%%(day).2d_%%(hour).2d%%(minute).2d-%%(rev)s_%%(slug)s
# sys.path path, will be prepended to sys.path if present.
# defaults to the current working directory.
prepend_sys_path = .
# timezone to use when rendering the date within the migration file
# as well as the filename.
# If specified, requires the python>=3.9 or backports.zoneinfo library.
# Any required deps can installed by adding `alembic[tz]` to the pip requirements
# string value is passed to ZoneInfo()
# leave blank for localtime
# timezone =
# max length of characters to apply to the "slug" field
# truncate_slug_length = 40
# set to 'true' to run the environment during
# the 'revision' command, regardless of autogenerate
# revision_environment = false
# set to 'true' to allow .pyc and .pyo files without
# a source .py file to be detected as revisions in the
# versions/ directory
# sourceless = false
# version location specification; This defaults
# to migrations/versions. When using multiple version
# directories, initial revisions must be specified with --version-path.
# The path separator used here should be the separator specified by "version_path_separator" below.
# version_locations = %(here)s/bar:%(here)s/bat:migrations/versions
# version path separator; As mentioned above, this is the character used to split
# version_locations. The default within new alembic.ini files is "os", which uses os.pathsep.
# If this key is omitted entirely, it falls back to the legacy behavior of splitting on spaces and/or commas.
# Valid values for version_path_separator are:
#
# version_path_separator = :
# version_path_separator = ;
# version_path_separator = space
# version_path_separator = newline
version_path_separator = os # Use os.pathsep. Default configuration used for new projects.
# set to 'true' to search source files recursively
# in each "version_locations" directory
# new in Alembic version 1.10
# recursive_version_locations = false
# the output encoding used when revision files
# are written from script.py.mako
# output_encoding = utf-8
sqlalchemy.url = driver://user:pass@localhost/dbname
[post_write_hooks]
# post_write_hooks defines scripts or Python functions that are run
# on newly generated revision scripts. See the documentation for further
# detail and examples
# format using "black" - use the console_scripts runner, against the "black" entrypoint
# hooks = black
# black.type = console_scripts
# black.entrypoint = black
# black.options = -l 79 REVISION_SCRIPT_FILENAME
# lint with attempts to fix using "ruff" - use the exec runner, execute a binary
# hooks = ruff
# ruff.type = exec
# ruff.executable = %(here)s/.venv/bin/ruff
# ruff.options = --fix REVISION_SCRIPT_FILENAME
# Logging configuration
[loggers]
keys = root,sqlalchemy,alembic
[handlers]
keys = console
[formatters]
keys = generic
[logger_root]
level = WARNING
handlers = console
qualname =
[logger_sqlalchemy]
level = WARNING
handlers =
qualname = sqlalchemy.engine
[logger_alembic]
level = INFO
handlers =
qualname = alembic
[handler_console]
class = StreamHandler
args = (sys.stderr,)
level = NOTSET
formatter = generic
[formatter_generic]
format = %(levelname)-5.5s [%(name)s] %(message)s
datefmt = %H:%M:%S

View File

@ -15,13 +15,13 @@ kill_timeout = '5s'
internal_port = 8000
auto_stop_machines = 'off'
auto_start_machines = true
min_machines_running = 1
min_machines_running = 3
processes = ['api']
[http_service.concurrency]
type = 'requests'
hard_limit = 250
soft_limit = 200
hard_limit = 50
soft_limit = 20
[[vm]]
memory = '512mb'

1
migrations/README Normal file
View File

@ -0,0 +1 @@
Generic single-database configuration.

95
migrations/env.py Normal file
View File

@ -0,0 +1,95 @@
import os
import sys
from logging.config import fileConfig
from pathlib import Path
from alembic import context
from dotenv import load_dotenv
from sqlalchemy import engine_from_config, pool
# Import your models
from src.db import Base
# Add project root to Python path
sys.path.append(str(Path(__file__).parents[1]))
# Load environment variables
load_dotenv()
# this is the Alembic Config object, which provides
# access to the values within the .ini file in use.
config = context.config
# Interpret the config file for Python logging.
# This line sets up loggers basically.
if config.config_file_name is not None:
fileConfig(config.config_file_name, disable_existing_loggers=False)
# add your model's MetaData object here
# for 'autogenerate' support
# from myapp import mymodel
# target_metadata = mymodel.Base.metadata
target_metadata = Base.metadata
# other values from the config, defined by the needs of env.py,
# can be acquired:
# my_important_option = config.get_main_option("my_important_option")
# ... etc.
def get_url():
return os.getenv("CONNECTION_URI")
def run_migrations_offline() -> None:
"""Run migrations in 'offline' mode.
This configures the context with just a URL
and not an Engine, though an Engine is acceptable
here as well. By skipping the Engine creation
we don't even need a DBAPI to be available.
Calls to context.execute() here emit the given string to the
script output.
"""
# url = config.get_main_option("sqlalchemy.url")
url = get_url()
context.configure(
url=url,
target_metadata=target_metadata,
literal_binds=True,
dialect_opts={"paramstyle": "named"},
)
with context.begin_transaction():
context.run_migrations()
def run_migrations_online() -> None:
"""Run migrations in 'online' mode.
In this scenario we need to create an Engine
and associate a connection with the context.
"""
configuration = config.get_section(config.config_ini_section)
configuration["sqlalchemy.url"] = get_url()
connectable = engine_from_config(
configuration,
prefix="sqlalchemy.",
poolclass=pool.NullPool,
)
with connectable.connect() as connection:
context.configure(connection=connection, target_metadata=target_metadata)
with context.begin_transaction():
context.run_migrations()
if context.is_offline_mode():
run_migrations_offline()
else:
run_migrations_online()

26
migrations/script.py.mako Normal file
View File

@ -0,0 +1,26 @@
"""${message}
Revision ID: ${up_revision}
Revises: ${down_revision | comma,n}
Create Date: ${create_date}
"""
from typing import Sequence, Union
from alembic import op
import sqlalchemy as sa
${imports if imports else ""}
# revision identifiers, used by Alembic.
revision: str = ${repr(up_revision)}
down_revision: Union[str, None] = ${repr(down_revision)}
branch_labels: Union[str, Sequence[str], None] = ${repr(branch_labels)}
depends_on: Union[str, Sequence[str], None] = ${repr(depends_on)}
def upgrade() -> None:
${upgrades if upgrades else "pass"}
def downgrade() -> None:
${downgrades if downgrades else "pass"}

View File

@ -0,0 +1,60 @@
"""Add indexes for messages and metamessages for reads
Revision ID: c3828084f472
Revises:
Create Date: 2024-12-12 13:41:40.156095
"""
from typing import Sequence, Union
import sqlalchemy as sa
from sqlalchemy import text
from alembic import op
from sqlalchemy.dialects import postgresql
# revision identifiers, used by Alembic.
revision: str = "c3828084f472"
down_revision: Union[str, None] = None
branch_labels: Union[str, Sequence[str], None] = None
depends_on: Union[str, Sequence[str], None] = None
def upgrade() -> None:
# Add new indexes
op.create_index("idx_users_app_lookup", "users", ["app_id", "public_id"])
op.create_index("idx_sessions_user_lookup", "sessions", ["user_id", "public_id"])
op.create_index(
"idx_messages_session_lookup",
"messages",
["session_id", "id"],
postgresql_include=[
"public_id",
"is_user",
"content",
"metadata",
"created_at",
],
)
op.create_index(
"idx_metamessages_lookup",
"metamessages",
["metamessage_type", sa.text("id DESC")],
postgresql_include=[
"public_id",
"content",
"message_id",
"created_at",
"metadata",
],
)
def downgrade() -> None:
# Remove new indexes
op.drop_index("idx_users_app_lookup", table_name="users")
op.drop_index("idx_sessions_user_lookup", table_name="sessions")
op.drop_index("idx_messages_session_lookup", table_name="messages")
op.drop_index("idx_metamessages_lookup", table_name="metamessages")

View File

@ -13,7 +13,7 @@ dependencies = [
"sqlalchemy>=2.0.30",
"fastapi-pagination>=0.12.24",
"pgvector>=0.2.5",
"sentry-sdk[fastapi,sqlalchemy]>=2.3.1",
"sentry-sdk[fastapi,sqlalchemy,anthropic]>=2.3.1",
"greenlet>=3.0.3",
"psycopg[binary]>=3.1.19",
"httpx>=0.27.0",
@ -21,6 +21,7 @@ dependencies = [
"openai>=1.43.0",
"anthropic>=0.36.0",
"nanoid>=2.0.0",
"alembic>=1.14.0",
]
[tool.uv]
dev-dependencies = [

View File

@ -2,8 +2,10 @@ import asyncio
import os
from collections.abc import Iterable
import sentry_sdk
from anthropic import Anthropic, MessageStreamManager
from dotenv import load_dotenv
from sentry_sdk.ai.monitoring import ai_track
from sqlalchemy import select
from sqlalchemy.ext.asyncio import AsyncSession
@ -37,45 +39,53 @@ class Dialectic:
self.chat_history = chat_history
self.client = Anthropic(api_key=os.getenv("ANTHROPIC_API_KEY"))
@ai_track("Dialectic Call")
def call(self):
prompt = f"""
You are tasked with responding to the query based on the context provided.
<query>{self.agent_input}</query>
<context>{self.user_representation}</context>
<conversation_history>{self.chat_history}</conversation_history>
Provide a brief, matter-of-fact, and appropriate response to the query based on the context provided. If the context provided doesn't aid in addressing the query, return only the word "None".
"""
with sentry_sdk.start_transaction(
op="dialectic-inference", name="Dialectic API Response"
):
prompt = f"""
You are tasked with responding to the query based on the context provided.
<query>{self.agent_input}</query>
<context>{self.user_representation}</context>
<conversation_history>{self.chat_history}</conversation_history>
Provide a brief, matter-of-fact, and appropriate response to the query based on the context provided. If the context provided doesn't aid in addressing the query, return only the word "None".
"""
response = self.client.messages.create(
messages=[
{
"role": "user",
"content": prompt,
}
],
model="claude-3-5-sonnet-20240620",
max_tokens=300,
)
return response.content
response = self.client.messages.create(
messages=[
{
"role": "user",
"content": prompt,
}
],
model="claude-3-5-sonnet-20240620",
max_tokens=300,
)
return response.content
@ai_track("Dialectic Call")
def stream(self):
prompt = f"""
You are tasked with responding to the query based on the context provided.
<query>{self.agent_input}</query>
<context>{self.user_representation}</context>
<conversation_history>{self.chat_history}</conversation_history>
Provide a brief, matter-of-fact, and appropriate response to the query based on the context provided. If the context provided doesn't aid in addressing the query, return only the word "None".
"""
return self.client.messages.stream(
model="claude-3-5-sonnet-20240620",
messages=[
{
"role": "user",
"content": prompt,
}
],
max_tokens=300,
)
with sentry_sdk.start_transaction(
op="dialectic-inference", name="Dialect API Response"
):
prompt = f"""
You are tasked with responding to the query based on the context provided.
<query>{self.agent_input}</query>
<context>{self.user_representation}</context>
<conversation_history>{self.chat_history}</conversation_history>
Provide a brief, matter-of-fact, and appropriate response to the query based on the context provided. If the context provided doesn't aid in addressing the query, return only the word "None".
"""
return self.client.messages.stream(
model="claude-3-5-sonnet-20241022",
messages=[
{
"role": "user",
"content": prompt,
}
],
max_tokens=300,
)
async def chat_history(app_id: str, user_id: str, session_id: str) -> str:

View File

@ -1,4 +1,3 @@
import datetime
from collections.abc import Sequence
from typing import Optional
@ -7,6 +6,7 @@ from openai import OpenAI
from sqlalchemy import Select, cast, insert, select
from sqlalchemy.exc import IntegrityError
from sqlalchemy.ext.asyncio import AsyncSession
from sqlalchemy.sql import func
from sqlalchemy.types import BigInteger
from . import models, schemas
@ -128,9 +128,9 @@ async def get_users(
stmt = stmt.where(models.User.h_metadata.contains(filter))
if reverse:
stmt = stmt.order_by(models.User.created_at.desc())
stmt = stmt.order_by(models.User.id.desc())
else:
stmt = stmt.order_by(models.User.created_at)
stmt = stmt.order_by(models.User.id)
return stmt
@ -205,9 +205,9 @@ async def get_sessions(
stmt = stmt.where(models.Session.h_metadata.contains(filter))
if reverse:
stmt = stmt.order_by(models.Session.created_at.desc())
stmt = stmt.order_by(models.Session.id.desc())
else:
stmt = stmt.order_by(models.Session.created_at)
stmt = stmt.order_by(models.Session.id)
return stmt
@ -443,9 +443,9 @@ async def get_messages(
stmt = stmt.where(models.Message.h_metadata.contains(filter))
if reverse:
stmt = stmt.order_by(models.Message.created_at.desc())
stmt = stmt.order_by(models.Message.id.desc())
else:
stmt = stmt.order_by(models.Message.created_at)
stmt = stmt.order_by(models.Message.id)
return stmt
@ -561,9 +561,9 @@ async def get_metamessages(
stmt = stmt.where(models.Metamessage.h_metadata.contains(filter))
if reverse:
stmt = stmt.order_by(models.Metamessage.created_at.desc())
stmt = stmt.order_by(models.Metamessage.id.desc())
else:
stmt = stmt.order_by(models.Metamessage.created_at)
stmt = stmt.order_by(models.Metamessage.id)
return stmt
@ -647,9 +647,9 @@ async def get_collections(
stmt = stmt.where(models.Collection.h_metadata.contains(filter))
if reverse:
stmt = stmt.order_by(models.Collection.created_at.desc())
stmt = stmt.order_by(models.Collection.id.desc())
else:
stmt = stmt.order_by(models.Collection.created_at)
stmt = stmt.order_by(models.Collection.id)
return stmt
@ -784,9 +784,9 @@ async def get_documents(
stmt = stmt.where(models.Document.h_metadata.contains(filter))
if reverse:
stmt = stmt.order_by(models.Document.created_at.desc())
stmt = stmt.order_by(models.Document.id.desc())
else:
stmt = stmt.order_by(models.Document.created_at)
stmt = stmt.order_by(models.Document.id)
return stmt
@ -906,7 +906,7 @@ async def update_document(
)
embedding = response.data[0].embedding
honcho_document.embedding = embedding
honcho_document.created_at = datetime.datetime.utcnow()
honcho_document.created_at = func.now()
if document.metadata is not None:
honcho_document.h_metadata = document.metadata

View File

@ -1,7 +1,9 @@
import os
from alembic import command
from alembic.config import Config
from dotenv import load_dotenv
from sqlalchemy import MetaData, create_engine
from sqlalchemy import MetaData
from sqlalchemy.ext.asyncio import async_sessionmaker, create_async_engine
from sqlalchemy.orm import declarative_base
@ -21,10 +23,15 @@ engine = create_async_engine(
connect_args=connect_args,
echo=True,
pool_pre_ping=True,
pool_size=20,
max_overflow=50,
)
SessionLocal = async_sessionmaker(
autocommit=False, autoflush=False, expire_on_commit=False, bind=engine
autocommit=False,
autoflush=False,
expire_on_commit=False,
bind=engine,
)
table_schema = os.getenv("DATABASE_SCHEMA")
@ -38,11 +45,5 @@ def scaffold_db():
"""use a sync engine for scaffolding the database. ddl operations are unavailable
with async engines
"""
print(os.environ["CONNECTION_URI"])
engine = create_engine(
os.environ["CONNECTION_URI"],
pool_pre_ping=True,
echo=True,
)
Base.metadata.create_all(bind=engine)
engine.dispose()
alembic_cfg = Config("alembic.ini")
command.upgrade(alembic_cfg, "head")

File diff suppressed because one or more lines are too long

View File

@ -5,6 +5,8 @@ import sentry_sdk
from fastapi import APIRouter, FastAPI
from fastapi.middleware.cors import CORSMiddleware
from fastapi_pagination import add_pagination
from sentry_sdk.integrations.fastapi import FastApiIntegration
from sentry_sdk.integrations.starlette import StarletteIntegration
from src.routers import (
apps,
@ -27,6 +29,14 @@ if SENTRY_ENABLED:
enable_tracing=True,
traces_sample_rate=0.4,
profiles_sample_rate=0.4,
integrations=[
StarletteIntegration(
transaction_style="endpoint",
),
FastApiIntegration(
transaction_style="endpoint",
),
],
)

View File

@ -50,7 +50,6 @@ async def create_app(app: schemas.AppCreate, db=db):
@router.get("/get_or_create/{name}", response_model=schemas.App)
async def get_or_create_app(name: str, db=db):
"""Get or Create an App"""
print("name", name)
app = await crud.get_app_by_name(db=db, name=name)
if app is None:
app = await create_app(db=db, app=schemas.AppCreate(name=name))

37
uv.lock
View File

@ -5,6 +5,20 @@ resolution-markers = [
"python_full_version >= '3.13'",
]
[[package]]
name = "alembic"
version = "1.14.0"
source = { registry = "https://pypi.org/simple" }
dependencies = [
{ name = "mako" },
{ name = "sqlalchemy" },
{ name = "typing-extensions" },
]
sdist = { url = "https://files.pythonhosted.org/packages/00/1e/8cb8900ba1b6360431e46fb7a89922916d3a1b017a8908a7c0499cc7e5f6/alembic-1.14.0.tar.gz", hash = "sha256:b00892b53b3642d0b8dbedba234dbf1924b69be83a9a769d5a624b01094e304b", size = 1916172 }
wheels = [
{ url = "https://files.pythonhosted.org/packages/cb/06/8b505aea3d77021b18dcbd8133aa1418f1a1e37e432a465b14c46b2c0eaa/alembic-1.14.0-py3-none-any.whl", hash = "sha256:99bd884ca390466db5e27ffccff1d179ec5c05c965cfefc0607e69f9e411cb25", size = 233482 },
]
[[package]]
name = "annotated-types"
version = "0.7.0"
@ -425,9 +439,10 @@ wheels = [
[[package]]
name = "honcho"
version = "0.0.14"
version = "0.0.15"
source = { virtual = "." }
dependencies = [
{ name = "alembic" },
{ name = "anthropic" },
{ name = "fastapi", extra = ["standard"] },
{ name = "fastapi-pagination" },
@ -439,7 +454,7 @@ dependencies = [
{ name = "psycopg", extra = ["binary"] },
{ name = "python-dotenv" },
{ name = "rich" },
{ name = "sentry-sdk", extra = ["fastapi", "sqlalchemy"] },
{ name = "sentry-sdk", extra = ["anthropic", "fastapi", "sqlalchemy"] },
{ name = "sqlalchemy" },
]
@ -455,6 +470,7 @@ dev = [
[package.metadata]
requires-dist = [
{ name = "alembic", specifier = ">=1.14.0" },
{ name = "anthropic", specifier = ">=0.36.0" },
{ name = "fastapi", extras = ["standard"], specifier = ">=0.111.0" },
{ name = "fastapi-pagination", specifier = ">=0.12.24" },
@ -466,7 +482,7 @@ requires-dist = [
{ name = "psycopg", extras = ["binary"], specifier = ">=3.1.19" },
{ name = "python-dotenv", specifier = ">=1.0.0" },
{ name = "rich", specifier = ">=13.7.1" },
{ name = "sentry-sdk", extras = ["fastapi", "sqlalchemy"], specifier = ">=2.3.1" },
{ name = "sentry-sdk", extras = ["fastapi", "sqlalchemy", "anthropic"], specifier = ">=2.3.1" },
{ name = "sqlalchemy", specifier = ">=2.0.30" },
]
@ -685,6 +701,18 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/b4/ee/6d9873144f860391fd1130be0e1e5a1dbd7e9d128da1c7baf1ae71babb99/jiter-0.6.1-cp39-none-win_amd64.whl", hash = "sha256:d465db62d2d10b489b7e7a33027c4ae3a64374425d757e963f86df5b5f2e7fc5", size = 202278 },
]
[[package]]
name = "mako"
version = "1.3.8"
source = { registry = "https://pypi.org/simple" }
dependencies = [
{ name = "markupsafe" },
]
sdist = { url = "https://files.pythonhosted.org/packages/5f/d9/8518279534ed7dace1795d5a47e49d5299dd0994eed1053996402a8902f9/mako-1.3.8.tar.gz", hash = "sha256:577b97e414580d3e088d47c2dbbe9594aa7a5146ed2875d4dfa9075af2dd3cc8", size = 392069 }
wheels = [
{ url = "https://files.pythonhosted.org/packages/1e/bf/7a6a36ce2e4cafdfb202752be68850e22607fccd692847c45c1ae3c17ba6/Mako-1.3.8-py3-none-any.whl", hash = "sha256:42f48953c7eb91332040ff567eb7eea69b22e7a4affbc5ba8e845e8f730f6627", size = 78569 },
]
[[package]]
name = "markdown-it-py"
version = "3.0.0"
@ -1239,6 +1267,9 @@ wheels = [
]
[package.optional-dependencies]
anthropic = [
{ name = "anthropic" },
]
fastapi = [
{ name = "fastapi" },
]