2.0.0 Alpha: Data Refinery
This commit is contained in:
@@ -0,0 +1,186 @@
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
import re
|
||||
import sqlite3
|
||||
from typing import Any, Dict
|
||||
|
||||
from database.job_store import JobStore
|
||||
from database.paths import L1_DB, OUTPUT_ARENA
|
||||
from web.config import Config
|
||||
|
||||
|
||||
MATCH_ID_PATTERN = re.compile(r'\bg161-[0-9]{10,}\b')
|
||||
|
||||
|
||||
class ImportValidationError(ValueError):
|
||||
pass
|
||||
|
||||
|
||||
class DuplicateMatchError(ImportValidationError):
|
||||
pass
|
||||
|
||||
|
||||
class MatchImportService:
|
||||
@staticmethod
|
||||
def validate_capture(raw_bytes: bytes) -> Dict[str, Any]:
|
||||
if not raw_bytes:
|
||||
raise ImportValidationError('Uploaded file is empty')
|
||||
|
||||
try:
|
||||
text = raw_bytes.decode('utf-8-sig')
|
||||
except UnicodeDecodeError as exc:
|
||||
raise ImportValidationError('Capture must be UTF-8 JSON') from exc
|
||||
|
||||
try:
|
||||
capture = json.loads(text)
|
||||
except json.JSONDecodeError as exc:
|
||||
raise ImportValidationError(
|
||||
f'Invalid JSON at line {exc.lineno}, column {exc.colno}'
|
||||
) from exc
|
||||
|
||||
if not isinstance(capture, list) or not capture:
|
||||
raise ImportValidationError(
|
||||
'Capture root must be a non-empty list of network responses'
|
||||
)
|
||||
|
||||
urls = []
|
||||
successful_responses = 0
|
||||
for index, item in enumerate(capture):
|
||||
if not isinstance(item, dict):
|
||||
raise ImportValidationError(
|
||||
f'Capture item {index} must be an object'
|
||||
)
|
||||
url = item.get('url')
|
||||
if not isinstance(url, str) or not url:
|
||||
raise ImportValidationError(
|
||||
f'Capture item {index} has no URL'
|
||||
)
|
||||
urls.append(url)
|
||||
if item.get('status') == 200 and item.get('body') is not None:
|
||||
successful_responses += 1
|
||||
|
||||
match_ids = sorted({
|
||||
match.group(0)
|
||||
for url in urls
|
||||
for match in MATCH_ID_PATTERN.finditer(url)
|
||||
})
|
||||
if len(match_ids) != 1:
|
||||
raise ImportValidationError(
|
||||
f'Capture must reference exactly one match ID; found {match_ids}'
|
||||
)
|
||||
if successful_responses < 2:
|
||||
raise ImportValidationError(
|
||||
'Capture does not contain enough successful API responses'
|
||||
)
|
||||
|
||||
match_id = match_ids[0]
|
||||
has_match_data = any(
|
||||
f'/api/data/match/{match_id}' in url for url in urls
|
||||
)
|
||||
has_round_data = any(
|
||||
f'/api/match/round/{match_id}' in url for url in urls
|
||||
)
|
||||
if not has_match_data or not has_round_data:
|
||||
missing = []
|
||||
if not has_match_data:
|
||||
missing.append('match data')
|
||||
if not has_round_data:
|
||||
missing.append('round data')
|
||||
raise ImportValidationError(
|
||||
f"Capture is missing required endpoint(s): {', '.join(missing)}"
|
||||
)
|
||||
|
||||
return {
|
||||
'match_id': match_id,
|
||||
'content_sha256': hashlib.sha256(raw_bytes).hexdigest(),
|
||||
'response_count': len(capture),
|
||||
'successful_responses': successful_responses,
|
||||
'text': text,
|
||||
}
|
||||
|
||||
@staticmethod
|
||||
def _existing_l1_content(match_id: str):
|
||||
if not L1_DB.exists():
|
||||
return None
|
||||
db = sqlite3.connect(str(L1_DB))
|
||||
try:
|
||||
row = db.execute(
|
||||
"""
|
||||
SELECT content
|
||||
FROM raw_iframe_network
|
||||
WHERE match_id = ?
|
||||
""",
|
||||
[match_id],
|
||||
).fetchone()
|
||||
return row[0] if row else None
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
@staticmethod
|
||||
def prepare_import(
|
||||
raw_bytes: bytes,
|
||||
original_filename: str,
|
||||
created_by: str,
|
||||
replace: bool = False,
|
||||
):
|
||||
validation = MatchImportService.validate_capture(raw_bytes)
|
||||
match_id = validation['match_id']
|
||||
content_hash = validation['content_sha256']
|
||||
|
||||
existing_content = MatchImportService._existing_l1_content(match_id)
|
||||
if existing_content is not None:
|
||||
existing_hash = hashlib.sha256(
|
||||
existing_content.encode('utf-8')
|
||||
).hexdigest()
|
||||
if existing_hash == content_hash:
|
||||
raise DuplicateMatchError(
|
||||
f'Match {match_id} is already imported with identical data'
|
||||
)
|
||||
if not replace:
|
||||
raise DuplicateMatchError(
|
||||
f'Match {match_id} already exists with different data; '
|
||||
'explicit replacement is required'
|
||||
)
|
||||
|
||||
match_dir = OUTPUT_ARENA / match_id
|
||||
match_dir.mkdir(parents=True, exist_ok=True)
|
||||
destination = match_dir / 'iframe_network.json'
|
||||
if destination.exists() and not replace:
|
||||
current_hash = hashlib.sha256(destination.read_bytes()).hexdigest()
|
||||
if current_hash == content_hash:
|
||||
raise DuplicateMatchError(
|
||||
f'Match {match_id} is already queued with identical data'
|
||||
)
|
||||
raise DuplicateMatchError(
|
||||
f'Pending capture already exists for {match_id}'
|
||||
)
|
||||
|
||||
temporary = destination.with_suffix('.json.tmp')
|
||||
temporary.write_bytes(raw_bytes)
|
||||
os.replace(str(temporary), str(destination))
|
||||
|
||||
store = JobStore(Config.DB_WEB_PATH)
|
||||
job_id = store.create_job(
|
||||
'match_import',
|
||||
match_id=match_id,
|
||||
input_path=str(destination),
|
||||
created_by=created_by,
|
||||
)
|
||||
store.upsert_match_import(
|
||||
match_id,
|
||||
content_hash,
|
||||
str(destination),
|
||||
'queued',
|
||||
job_id,
|
||||
)
|
||||
return {
|
||||
'job_id': job_id,
|
||||
'match_id': match_id,
|
||||
'content_sha256': content_hash,
|
||||
'response_count': validation['response_count'],
|
||||
'source_path': str(destination),
|
||||
'original_filename': Path(original_filename or '').name,
|
||||
'replace': bool(replace),
|
||||
}
|
||||
Reference in New Issue
Block a user