Skip to content
Merged
Show file tree
Hide file tree
Changes from 12 commits
Commits
Show all changes
18 commits
Select commit Hold shift + click to select a range
bfb245a
chore(types): Just check everything
MorganBennetDev Sep 25, 2026
0e3b689
chore(types): Baseline update
MorganBennetDev Sep 28, 2026
0cabd6c
fix(types): Use overload to correctly type `find_docket_object`
MorganBennetDev Sep 28, 2026
22f1831
fix(types): `missing-attribute` caused by field blending in
MorganBennetDev Sep 28, 2026
c2a5999
fix(types): Manager `missing-attribute` errors
MorganBennetDev Sep 28, 2026
8c2954d
fix(types): Used TypedDict to fix errors in
MorganBennetDev Oct 2, 2026
c7ed95f
fix(types): Most `no-matching-overload` errors
MorganBennetDev Oct 2, 2026
6f2f086
fix(types): Undo breaking `FederalCourtsManager` and
MorganBennetDev Oct 2, 2026
b879211
chore(types): Post-rebase baseline update
MorganBennetDev Oct 6, 2026
3b59e20
chore(corpus_importer): Delete unused command and module
MorganBennetDev Oct 6, 2026
8cb2a3f
chore(types): Address review comment
MorganBennetDev Oct 6, 2026
2736317
fix(audio): Failing test
MorganBennetDev Oct 6, 2026
376bc01
chore(types): Move ClusterCitation and Opinion from query sets to
MorganBennetDev Oct 7, 2026
d054eed
Merge branch 'main' into morgan/ci-typeck-fixes-2
MorganBennetDev Oct 7, 2026
f66c692
fix(types): Bad baseline merge
MorganBennetDev Oct 7, 2026
d1a9ff4
revert(types): Restore ClusterCitation and Opinion query sets
MorganBennetDev Oct 7, 2026
f6be5e8
revert(types): Restore FederalCourts and StateCourts query sets
MorganBennetDev Oct 7, 2026
808ed9c
Merge branch 'main' into morgan/ci-typeck-fixes-2
albertisfu Oct 8, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6,431 changes: 2,246 additions & 4,185 deletions .pyrefly-baseline.json

Large diffs are not rendered by default.

18 changes: 8 additions & 10 deletions cl/ai/management/commands/send_gemini_file_batches.py
Original file line number Diff line number Diff line change
Expand Up @@ -191,22 +191,20 @@ def get_s3_file_list(

# Create S3 client with session token support for dev mode
try:
client_kwargs = {
"service_name": "s3",
"aws_access_key_id": key_id,
"aws_secret_access_key": secret,
}

aws_session_token = None
# In dev mode, add session token if available
if settings.DEVELOPMENT:
env = environ.FileAwareEnv()
session_token = env("AWS_SESSION_TOKEN", default=None) or env(
aws_session_token = env("AWS_SESSION_TOKEN", default=None) or env(
"AWS_DEV_SESSION_TOKEN", default=None
)
if session_token:
client_kwargs["aws_session_token"] = session_token

s3_client = boto3.client(**client_kwargs)
s3_client = boto3.client(
"s3",
aws_access_key_id=key_id,
aws_secret_access_key=secret,
aws_session_token=aws_session_token,
)
except (BotoCoreError, ClientError) as e:
raise CommandError(f"Failed to create S3 client: {e}")

Expand Down
7 changes: 1 addition & 6 deletions cl/api/pagination.py
Original file line number Diff line number Diff line change
Expand Up @@ -326,19 +326,14 @@ def get_paginated_response(

base_response = {
"count": self.get_results_count(),
}
remaining_fields = {
"next": self.get_next_link(cached_response),
"previous": self.get_previous_link(cached_response),
"results": data,
}

if self.search_type == SEARCH_TYPES.RECAP:
# Include the document_count for the "r" search type.
base_response.update(
{"document_count": self.get_child_results_count()}
)
base_response.update(remaining_fields)
Comment thread
MorganBennetDev marked this conversation as resolved.
base_response["document_count"] = self.get_child_results_count()
return Response(base_response)

def get_next_link(self, cached_response: bool) -> str | None:
Expand Down
1 change: 1 addition & 0 deletions cl/api/utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -2186,6 +2186,7 @@ def fields(self):
allowed = set(filter(None, filter_fields))

# omit fields in the `omit` argument.
# TODO: Typing this correctly involves going through a long chain of things that probably also need to be typed
omitted = set(filter(None, omit_fields))

for field in existing:
Expand Down
36 changes: 17 additions & 19 deletions cl/audio/tasks.py
Original file line number Diff line number Diff line change
Expand Up @@ -21,6 +21,7 @@
OpenAI,
RateLimitError,
UnprocessableEntityError,
omit,
)
from sentry_sdk import capture_exception

Expand Down Expand Up @@ -152,23 +153,21 @@ def transcribe_from_open_ai_api(self, audio_pk: int, dont_retry: bool = False):

# Prevent default openai client retrying
client = stack.enter_context(OpenAI(max_retries=0))
kwargs = {
"file": file,
"model": "whisper-1",
"language": "en",
"response_format": "verbose_json",
"timestamp_granularities": ["word", "segment"],
"prompt": audio.case_name,
}

# The most common hallucination we have seen is the case name
# repeated in a loop. Manual testing showed that not sending
# the case name helps to get a clean transcript
if audio.stt_status == Audio.STT_HALLUCINATION:
kwargs.pop("prompt", "")

try:
transcript = client.audio.transcriptions.create(**kwargs)
transcript = client.audio.transcriptions.create(
file=file,
model="whisper-1",
language="en",
response_format="verbose_json",
timestamp_granularities=["word", "segment"],
# The most common hallucination we have seen is the case name
# repeated in a loop. Manual testing showed that not sending
# the case name helps to get a clean transcript
prompt=omit
if audio.stt_status == Audio.STT_HALLUCINATION
else audio.case_name,
)
except APIConnectionError as exc:
# Transient TCP / DNS blip. Usually resolves in seconds, so a
# short in-task retry is cheaper than waiting a full daemon
Expand Down Expand Up @@ -217,11 +216,9 @@ def transcribe_from_open_ai_api(self, audio_pk: int, dont_retry: bool = False):
capture_exception(e)
return

transcript_dict = transcript.to_dict()

with transaction.atomic():
audio.stt_transcript = transcript_dict["text"]
audio.duration = ceil(transcript_dict["duration"])
audio.stt_transcript = transcript.text
audio.duration = ceil(transcript.duration)
audio.stt_source = Audio.STT_OPENAI_WHISPER

if transcription_was_hallucinated(audio):
Expand All @@ -234,6 +231,7 @@ def transcribe_from_open_ai_api(self, audio_pk: int, dont_retry: bool = False):
audio.stt_status = Audio.STT_COMPLETE

audio.save()
transcript_dict = transcript.to_dict()
metadata = {
"segments": transcript_dict["segments"],
"words": transcript_dict["words"],
Expand Down
4 changes: 4 additions & 0 deletions cl/audio/tests.py
Original file line number Diff line number Diff line change
Expand Up @@ -527,6 +527,10 @@ def setUpTestData(cls) -> None:
}

class OpenAITranscription:
def __init__(self):
self.text = cls.open_ai_api_returned_dict["text"]
self.duration = cls.open_ai_api_returned_dict["duration"]

def to_dict(self):
return cls.open_ai_api_returned_dict

Expand Down
Empty file.
58 changes: 0 additions & 58 deletions cl/corpus_importer/import_columbia/analyze_import_exceptions.py

This file was deleted.

Loading
Loading