"""Assess explicitly selected Hume integration files locally (Python 3.10+, POSIX). No network, inference, imports of customer code, directory crawling, or exports. Only allowlisted findings are reported; source text, paths, credentials, audio, transcripts, labels and scores are never copied into the report. Code: MIT. See https://oruk.ai/migrate/hume for migration scope and the existing comparator. """ from __future__ import annotations import argparse import json import math import os from pathlib import Path import re import stat import sys MAX_FILES = 16 MAX_FILE_BYTES = 2 * 1024 * 1024 MAX_TOTAL_BYTES = 8 * 1024 * 1024 MAX_JSON_DEPTH = 24 MAX_JSON_NODES = 50_000 SCHEMA_VERSION = 1 REQUIREMENTS_KIND = "oruk.hume-migration.requirements" FEATURES = { "expression_audio", "speaking_style", "transcription", "interim_transcripts", "timed_segments", "speaker_labels", "tts", "voice_cloning", "tool_calls", "conversation_history", "interruptions", "face_analysis", "text_emotion", "vocal_bursts", "exact_hume_taxonomy", "tagger_dimensions", } FORMATS = {"wav", "flac", "mp3", "m4a", "ogg", "webm", "pcm_s16le", "pcm_f32le"} LANGUAGES = set("en bg hr cs da nl et fi fr de el hu it lv lt mt pl pt ro ru sk sl es sv uk tr ar hi ja ko vi no zh".split()) | {"other"} SPECTRA_LANGUAGES = set("bg hr cs da nl en et fi fr de el hu it lv lt mt pl pt ro ru sk sl es sv uk".split()) REALTIME_LANGUAGES = set("en es fr it pt nl de tr ru ar hi ja ko vi uk pl sv cs no da bg fi hr sk zh hu ro et".split()) AUDIO_FIELDS = {"formats", "channels", "sample_rate_hz", "max_clip_seconds", "max_file_bytes", "max_session_seconds"} # These are lexical clues, not proof that a path executes or that a customer exists. SOURCE_PATTERNS = { "legacy_batch_sdk": r"\bHumeBatchClient\b", "legacy_stream_sdk": r"\bHumeStreamClient\b", "expression_batch_namespace": r"\b(?:expression_measurement|expressionMeasurement)\s*\.\s*batch\b|/v0/batch/jobs\b", "expression_stream_namespace": r"\b(?:expression_measurement|expressionMeasurement)\s*\.\s*stream(?:ing)?\b", "evi": r"\bempathic_voice\b|\bempathicVoice\b|\bHumeVoiceClient\b|/v0/evi/", "tts": r"\bHumeTTSService\b|/v0/tts\b|\bhume\s*\.\s*tts\b", "current_tagger_reference": r"\bHume[ _-]+Tagger\b", "current_prosody_reference": r"\bHume[ _-]+Prosody\b", "hume_dependency_reference": r"\bfrom\s+hume(?:\.|\s)|\bimport\s+hume\b|[\"']hume[\"']|(?m:^\s*hume(?:\[legacy\])?\s*[=<>~!])", } class AssessmentError(ValueError): """Contains a fixed error code only, never untrusted input.""" def fail(code: str): raise AssessmentError(code) def parent_fd(path: Path): """Traverse without following symlinks, including parent components.""" if os.name != "posix" or not hasattr(os, "O_NOFOLLOW"): fail("posix_no_follow_required") full_path = os.fspath(path) if path.is_absolute() else os.path.join(os.getcwd(), os.fspath(path)) parts = Path(full_path).parts if ".." in parts: fail("parent_traversal_not_supported") fd = os.open(parts[0], os.O_RDONLY | os.O_DIRECTORY) try: for part in parts[1:-1]: next_fd = os.open(part, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW, dir_fd=fd) os.close(fd) fd = next_fd return fd, parts[-1] except OSError: os.close(fd) fail("unsafe_or_unreadable_path") def read_selected(path: Path, remaining=MAX_TOTAL_BYTES) -> bytes: fd, name = parent_fd(path) try: file_fd = os.open(name, os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK, dir_fd=fd) with os.fdopen(file_fd, "rb") as source: info = os.fstat(source.fileno()) if not stat.S_ISREG(info.st_mode): fail("regular_files_only") if info.st_size > MAX_FILE_BYTES: fail("file_byte_limit") if info.st_size > remaining: fail("total_byte_limit") data = source.read(min(MAX_FILE_BYTES, remaining) + 1) if len(data) > MAX_FILE_BYTES: fail("file_byte_limit") if len(data) > remaining: fail("total_byte_limit") return data except OSError: fail("unsafe_or_unreadable_file") finally: os.close(fd) def write_new(path: Path, data: str): fd, name = parent_fd(path) try: output_fd = os.open(name, os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW, 0o600, dir_fd=fd) with os.fdopen(output_fd, "w", encoding="utf-8") as output: output.write(data) except OSError: fail("output_exists_or_unwritable") finally: os.close(fd) def objects(value): pending = [(value, 0)] count = 0 while pending: item, depth = pending.pop() count += 1 if count > MAX_JSON_NODES: fail("json_node_limit") if depth > MAX_JSON_DEPTH: fail("json_depth_limit") yield item if isinstance(item, dict): pending.extend((child, depth + 1) for child in item.values()) elif isinstance(item, list): pending.extend((child, depth + 1) for child in item) def read_json(data: bytes): def pairs(items): result = {} for key, value in items: if key in result: fail("duplicate_json_key") result[key] = value return result try: value = json.loads(data.decode("utf-8"), object_pairs_hook=pairs, parse_constant=lambda _: fail("nonfinite_json_number")) for item in objects(value): if isinstance(item, float) and not math.isfinite(item): fail("nonfinite_json_number") return value except AssessmentError: raise except (ValueError, UnicodeError, RecursionError): fail("invalid_or_excessive_json") def enum_list(value, allowed): if not isinstance(value, list) or len(value) > len(allowed): fail("invalid_requirements") if any(not isinstance(item, str) or item not in allowed for item in value): fail("invalid_requirements") if len(set(value)) != len(value): fail("invalid_requirements") return sorted(value) def requirements(value): if not isinstance(value, dict) or set(value) - {"schema_version", "kind", "synthetic_example", "features", "languages", "audio"}: fail("invalid_requirements") if type(value.get("schema_version")) is not int or value["schema_version"] != SCHEMA_VERSION or value.get("kind") != REQUIREMENTS_KIND: fail("unsupported_requirements_schema") if "synthetic_example" in value and type(value["synthetic_example"]) is not bool: fail("invalid_requirements") audio = value.get("audio", {}) if not isinstance(audio, dict) or set(audio) - AUDIO_FIELDS: fail("invalid_requirements") clean_audio = {} for key in sorted(AUDIO_FIELDS): item = audio.get(key) if key == "formats": clean_audio[key] = enum_list(item or [], FORMATS) if item is None or isinstance(item, list) else fail("invalid_requirements") elif item is None: clean_audio[key] = None else: if type(item) not in (int, float) or not 0 < item <= 10**12 or not math.isfinite(item): fail("invalid_requirements") if key in {"channels", "sample_rate_hz", "max_file_bytes"} and type(item) is not int: fail("invalid_requirements") clean_audio[key] = item return {"schema_version": SCHEMA_VERSION, "kind": REQUIREMENTS_KIND, "synthetic_example": value.get("synthetic_example", False), "features": enum_list(value.get("features", []), FEATURES), "languages": enum_list(value.get("languages", []), LANGUAGES), "audio": clean_audio} def inspect_source(data: bytes): try: text = data.decode("utf-8") except UnicodeError: fail("source_must_be_utf8_text") if "\x00" in text: fail("source_must_be_utf8_text") codes = {code for code, pattern in SOURCE_PATTERNS.items() if re.search(pattern, text, re.I)} if "hume_dependency_reference" in codes and re.search(r"\b(?:client|hume)\s*\.\s*tts\b|livekit\.plugins\.hume", text): codes.add("tts") # Capture only short semantic versions associated with a Hume dependency. version_pattern = r"(?:[\"']hume[\"']\s*:\s*[\"'][~^]?|\bhume(?:\[legacy\])?\s*==\s*)(\d{1,4}\.\d{1,4}\.\d{1,4}(?:rc\d{1,3})?)(?![\w.])" versions = sorted(set(re.findall(version_pattern, text, re.I)))[:16] return sorted(codes), versions def inspect_archive(value): codes = set() for item in objects(value): if not isinstance(item, dict): continue models = item.get("models") prosody = models.get("prosody") if isinstance(models, dict) else None if item.get("type") == "user_message" and isinstance(item.get("message"), dict) and isinstance(models, dict): codes.add("evi_response_envelope") if isinstance(prosody, dict) and isinstance(prosody.get("scores"), dict): codes.add("evi_prosody_scores_present") results = item.get("results") predictions = results.get("predictions") if isinstance(results, dict) else None if isinstance(item.get("source"), dict) and isinstance(predictions, list): for prediction in predictions: if not isinstance(prediction, dict): continue model = prediction.get("models") prosody = model.get("prosody") if isinstance(model, dict) else None if isinstance(prosody, dict) and isinstance(prosody.get("grouped_predictions"), list): codes.add("legacy_batch_response_envelope") return sorted(codes) def candidate_routes(spec): """Documented native limits only; a candidate is never a parity guarantee.""" file_formats = {"wav", "flac", "mp3", "m4a", "ogg", "webm"} profiles = [ ("resonance_file", {"expression_audio", "speaking_style", "transcription", "timed_segments", "speaker_labels"}, file_formats, {"en"}, None, None, 3600, 30_000_000, None), ("resonance2_file", {"expression_audio", "speaking_style"}, file_formats, None, None, None, 120, 30 * 1024**2, None), ("spectra2_file", {"expression_audio", "speaking_style", "transcription"}, {"wav", "flac", "pcm_s16le", "pcm_f32le"}, SPECTRA_LANGUAGES, 1, (16000, 16000), 60, 4 * 1024**2, None), ("spectra2_stream", {"expression_audio", "speaking_style", "transcription"}, {"pcm_s16le"}, SPECTRA_LANGUAGES, 1, (16000, 16000), 60, None, 75), ("realtime", {"expression_audio", "transcription", "interim_transcripts", "timed_segments", "speaker_labels"}, {"pcm_s16le"}, REALTIME_LANGUAGES, 1, (8000, 96000), None, None, 600), ("resonance2_stream", {"expression_audio", "speaking_style"}, {"pcm_f32le"}, None, 1, (16000, 16000), 120, None, 600), ] output = [] audio = spec["audio"] for name, features, formats, languages, channels, rates, seconds, size, session in profiles: blockers = ["feature:" + item for item in sorted(set(spec["features"]) - features)] unknowns = [] if not spec["features"]: unknowns.append("required_features") if not spec["languages"]: unknowns.append("languages") if languages is None: unknowns.append("language_expression_performance") elif set(spec["languages"]) - languages: blockers.append("language_not_listed") if "expression_audio" in spec["features"] or "speaking_style" in spec["features"]: unknowns.append("workload_expression_accuracy_and_thresholds") if not audio["formats"]: unknowns.append("input_formats") elif set(audio["formats"]) - formats: blockers.append("input_format_requires_conversion") for key, limit in [("max_clip_seconds", seconds), ("max_file_bytes", size), ("max_session_seconds", session)]: if audio[key] is None: unknowns.append(key) elif limit is None: unknowns.append(key + "_contract_not_assessed") elif audio[key] > limit: blockers.append(key + "_exceeds_limit") if audio["channels"] is None or channels is None: unknowns.append("channels_not_assessed") elif audio["channels"] != channels: blockers.append("channels_require_conversion") if audio["sample_rate_hz"] is None or rates is None: unknowns.append("sample_rate_not_assessed") elif not rates[0] <= audio["sample_rate_hz"] <= rates[1]: blockers.append("sample_rate_requires_conversion") output.append({"route": name, "status": "native_contract_mismatch" if blockers else "candidate_requires_validation", "blockers": blockers, "unknowns": sorted(set(unknowns))}) return output def assess(sources, archives, manifest_path=None): paths = [("source", path) for path in sources] + [("archive", path) for path in archives] if manifest_path is not None: paths.append(("requirements", manifest_path)) if not paths or len(paths) > MAX_FILES: fail("select_between_1_and_16_files") spec = requirements({"schema_version": 1, "kind": REQUIREMENTS_KIND}) inputs, all_codes, versions = [], set(), set() total = 0 for index, (kind, path) in enumerate(paths, 1): data = read_selected(path, MAX_TOTAL_BYTES - total) total += len(data) if total > MAX_TOTAL_BYTES: fail("total_byte_limit") codes, pins = [], [] if kind == "source": codes, pins = inspect_source(data) elif kind == "archive": codes = inspect_archive(read_json(data)) else: spec = requirements(read_json(data)) inputs.append({"input_id": f"input-{index}", "kind": kind, "bytes": len(data), "finding_codes": codes, "hume_version_pins": pins}) all_codes.update(codes) versions.update(pins) routes = [] if all_codes & {"legacy_batch_sdk", "expression_batch_namespace", "legacy_batch_response_envelope"}: routes.append("qualify_legacy_batch_audio_and_reuse_comparator") if all_codes & {"legacy_stream_sdk", "expression_stream_namespace"}: routes.append("qualify_legacy_stream_modality_and_protocol") if all_codes & {"evi", "evi_response_envelope"}: routes.append("partner_voice_stack_with_optional_oruk_expression") if "tts" in all_codes: routes.append("partner_tts_replacement_or_referral") if all_codes & {"current_tagger_reference", "current_prosody_reference"}: routes.append("confirm_current_contact_led_hume_contract") if not routes: routes.append("unknown_manual_qualification_required") return { "schema_version": SCHEMA_VERSION, "kind": "oruk.hume-migration.assessment", "tool_version": "1.0.0", "reviewed_on": "2026-10-03", "customer_or_active_usage_verified": False, "compatibility_verified": False, "input_inventory": inputs, "suggested_review_routes": routes, "migration_manifest": {"schema_version": 1, "kind": "oruk.hume-migration.manifest", "observed_findings": sorted(all_codes), "hume_version_pins": sorted(versions), "requirements": spec, "export_state": "not_performed"}, "native_route_assessment": candidate_routes(spec), "cutoff": {"product_scope": ["EVI", "TTS"], "utc": "2026-11-13T05:01:00Z", "us_pacific": "2026-11-12T21:01:00-08:00", "source": "https://dev.hume.ai/intro", "account_data_notice": "Hume states account data is permanently deleted after November 13, 2026."}, "limitations": [ "Lexical clues and JSON envelopes are not evidence of executing code, active use, customers, or valid scores.", "Source comments and example code can produce findings; inspect the selected files locally.", "A hume dependency alone does not identify the product or a migration customer.", "Current Hume Tagger/Prosody contracts need separate confirmation; the EVI/TTS cutoff does not prove their closure.", "Legacy Python interfaces changed at 0.7.0; the legacy extra last appeared in 0.8.6 and was removed in 0.9.0.", "Native route checks cover selected documented limits, not deployment capacity, latency, billing, model access, region, or production parity.", "Transcription language support does not prove expression accuracy. Never copy thresholds or fill missing labels with zero.", "EVI message times use milliseconds; legacy batch and Oruk timed outputs use seconds. Explicit alignment is required.", "Spectra-2 streaming uploads return one final result after finish, not interim words or automatic speech boundaries.", "Requirements are customer declarations. Unspecified fields remain unknown; no media is opened or measured.", ], "export_checklist": [ {"item": item, "status": "customer_to_verify"} for item in [ "Customer authorizes the account, lawful data scope, retention and destination before any export.", "Export configuration, prompt and tool versions, webhooks and session settings; omit secrets.", "Export chats, groups, paginated events and available audio into customer-controlled storage.", "Record counts, failures and checksums, resume incomplete exports, and verify restoration.", "Inventory voice metadata, lawful source recordings and permissions; voice IDs and weights are not portable.", "Target verified exports by November 5 and production cutover by November 9, 2026.", "Keep a validated non-Hume fallback after cutoff and confirm deletion/retention responsibilities.", ] ], "references": ["https://github.com/HumeAI/hume-python-sdk", "https://www.hume.ai/expression-measurement-api", "https://oruk.ai/docs", "https://oruk.ai/models", "https://oruk.ai/security", "https://oruk.ai/examples/compare-legacy-hume.py"], } class SafeParser(argparse.ArgumentParser): def error(self, message): self.exit(2, "Invalid arguments. Use --help; argument values are not echoed.\n") def main(argv=None): parser = SafeParser(description=__doc__) parser.add_argument("--source", action="append", type=Path, default=[], help="explicit UTF-8 source/lock/config file; repeatable") parser.add_argument("--archive", action="append", type=Path, default=[], help="explicit archived JSON response; repeatable") parser.add_argument("--requirements", type=Path, help="strict versioned requirements JSON; no free text or credentials") parser.add_argument("--output", required=True, type=Path, help="new local JSON report; never overwrite") args = parser.parse_args(argv) try: report = assess(args.source, args.archive, args.requirements) write_new(args.output, json.dumps(report, indent=2, allow_nan=False) + "\n") print("Saved local assessment. No active usage, compatibility, or exports were verified.") return 0 except AssessmentError as error: print("Assessment failed: " + str(error), file=sys.stderr) except (OSError, ValueError, TypeError, RecursionError, OverflowError): print("Assessment failed: invalid_or_inaccessible_input", file=sys.stderr) return 1 if __name__ == "__main__": raise SystemExit(main())