diff --git a/CHANGELOG.md b/CHANGELOG.md index bd0c7dd..e6988f3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,17 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +### Added + +- Assignment fields: `image_available_status_code`, `attorney_docket_number`, `domestic_representative` +- Address fields: `country_or_state_code`, `ict_state_code`, `ict_country_code` +- PCT application number format support in `sanitize_application_number()` + +### Changed + +- **BREAKING**: Assignment `correspondence_address_bag` changed to `correspondence_address` (single object, not list) +- All `PatentDataClient` methods now automatically sanitize application numbers before API requests + ## [0.2.1] ### Added diff --git a/examples/patent_data_example.py b/examples/patent_data_example.py index d9d114f..f77b0d3 100644 --- a/examples/patent_data_example.py +++ b/examples/patent_data_example.py @@ -203,6 +203,36 @@ else: print("No documents listed for this application.") + # Example: Download publication XML (grant or pgpub) + print("\nChecking for publication files (grant/pgpub XML)...") + if patent_wrapper_detail.grant_document_meta_data: + grant_metadata = patent_wrapper_detail.grant_document_meta_data + print(f"Grant document available: {grant_metadata.xml_file_name}") + print(f" Product: {grant_metadata.product_identifier}") + print(f" Created: {grant_metadata.file_create_date_time}") + + # Download grant XML to downloads folder with auto-generated filename + print("\nDownloading grant XML...") + grant_path = client.download_publication( + printed_metadata=grant_metadata, + destination_path="./download-example", + overwrite=True, + ) + print(f"Downloaded grant XML to: {grant_path}") + + if patent_wrapper_detail.pgpub_document_meta_data: + pgpub_metadata = patent_wrapper_detail.pgpub_document_meta_data + print(f"\nPre-grant publication available: {pgpub_metadata.xml_file_name}") + + # Download with custom filename + pgpub_path = client.download_publication( + printed_metadata=pgpub_metadata, + file_name="my_pgpub.xml", + destination_path="./download-example", + overwrite=True, + ) + print(f"Downloaded pgpub XML to: {pgpub_path}") + if patent_wrapper_detail.assignment_bag: print("\nAssignments:") for assignment in patent_wrapper_detail.assignment_bag: diff --git a/src/pyUSPTO/clients/base.py b/src/pyUSPTO/clients/base.py index 9193e93..73491f7 100644 --- a/src/pyUSPTO/clients/base.py +++ b/src/pyUSPTO/clients/base.py @@ -313,6 +313,44 @@ def _extract_filename_from_content_disposition( return None + @staticmethod + def _get_extension_from_mime_type(mime_type: Optional[str]) -> Optional[str]: + """Map MIME type to file extension. + + Maps common USPTO file formats to their appropriate extensions. + + Args: + mime_type: The MIME type from Content-Type header (e.g., "application/pdf"). + + Returns: + Optional[str]: File extension including dot (e.g., ".pdf"), or None if unmapped. + + Examples: + >>> _get_extension_from_mime_type("application/pdf") + '.pdf' + >>> _get_extension_from_mime_type("image/tiff") + '.tif' + >>> _get_extension_from_mime_type("unknown/type") + None + """ + if not mime_type: + return None + + # Normalize MIME type (remove charset and other parameters) + mime_type = mime_type.split(";")[0].strip().lower() + + # Map of common USPTO file MIME types to extensions + mime_to_ext = { + "application/pdf": ".pdf", + "image/tiff": ".tif", + "image/tif": ".tif", + "application/xml": ".xml", + "text/xml": ".xml", + "application/zip": ".zip", + } + + return mime_to_ext.get(mime_type) + def _save_response_to_file( self, response: requests.Response, file_path: str, overwrite: bool = False ) -> str: @@ -321,6 +359,9 @@ def _save_response_to_file( If file_path is a directory, attempts to extract filename from Content-Disposition header and save in that directory. + If file_path has no extension and Content-Disposition doesn't provide + a filename, attempts to determine extension from Content-Type header. + Args: response: Streaming response object from requests file_path: Local path where file should be saved. Can be a file path @@ -348,6 +389,16 @@ def _save_response_to_file( "header does not contain a filename. Please provide a full file path." ) path = path / filename + # If path has no extension, try to determine it from Content-Type + elif not path.suffix: + # Only attempt if Content-Disposition doesn't already provide filename + content_disp = response.headers.get("Content-Disposition") + if not self._extract_filename_from_content_disposition(content_disp): + # Try to get extension from Content-Type header + content_type = response.headers.get("Content-Type") + extension = self._get_extension_from_mime_type(content_type) + if extension: + path = path.with_suffix(extension) # Check for existing file if path.exists() and not overwrite: diff --git a/src/pyUSPTO/clients/patent_data.py b/src/pyUSPTO/clients/patent_data.py index ffba998..c988e58 100644 --- a/src/pyUSPTO/clients/patent_data.py +++ b/src/pyUSPTO/clients/patent_data.py @@ -75,6 +75,7 @@ def sanitize_application_number(self, input_number: str) -> str: Application numbers are either: - 8 digits (e.g., "16123456") - Series code format: 2 digits + "/" + 6 digits (e.g., "08/123456") + - PCT format: "PCT/US2024/012345" → "PCTUS2412345" This method removes common separators (commas, spaces) while preserving the "/" in series code format. @@ -102,8 +103,50 @@ def sanitize_application_number(self, input_number: str) -> str: if not input_number or not input_number.strip(): raise ValueError("Application number cannot be empty") + raw = input_number.strip() + + # --- NEW: Handle PCT formats --- + # Example: "PCT/US2024/012345" -> "PCTUS2412345" + if raw.startswith("PCT"): + parts = raw.split("/") + if len(parts) != 3: + raise ValueError( + f"Invalid PCT application format: {input_number}. " + "Expected PCT/CCYYYY/NNNNNN" + ) + + _, country_year, serial = parts + + # country_year can be "US2024" or "US24" + country = country_year[:2] + + year_part = country_year[2:] + if not year_part.isdigit(): + raise ValueError( + f"Invalid PCT year in: {country_year}. Must be digits." + ) + + # Normalize: + # "2024" -> "24" + # "24" -> "24" + if len(year_part) == 4: + year = year_part[-2:] + elif len(year_part) == 2: + year = year_part + else: + raise ValueError( + f"Invalid PCT year length in: {country_year}. " + "Expected CCYYYY or CCYY." + ) + + # Serial must be digits only + if not serial.isdigit(): + raise ValueError(f"Invalid PCT serial: {serial}. Must be numeric.") + + return f"PCT{country}{year}{serial}" + # Strip whitespace and remove commas/spaces - cleaned = input_number.strip().replace(",", "").replace(" ", "") + cleaned = raw.replace(",", "").replace(" ", "") # Check if this is series code format (NN/NNNNNN) if "/" in cleaned: @@ -157,7 +200,8 @@ def _get_wrapper_from_response( if ( application_number_for_validation - and wrapper.application_number_text != application_number_for_validation + and wrapper.application_number_text + != self.sanitize_application_number(application_number_for_validation) ): warnings.warn( f"API returned application number '{wrapper.application_number_text}' " @@ -433,7 +477,8 @@ def get_application_by_number( Args: application_number (str): The USPTO application number for the patent - application (e.g., "16123456"). + application (e.g., "16123456" or "18/915,708"). The application + number will be automatically sanitized to remove commas and spaces. Returns: Optional[PatentFileWrapper]: A `PatentFileWrapper` object representing @@ -445,7 +490,7 @@ def get_application_by_number( does not contain the expected data. """ endpoint = self.ENDPOINTS["get_application_by_number"].format( - application_number=application_number + application_number=self.sanitize_application_number(application_number) ) response_data = self._make_request( method="GET", endpoint=endpoint, response_class=PatentDataResponse @@ -469,7 +514,8 @@ def get_application_metadata( Args: application_number (str): The USPTO application number for which - metadata is being requested (e.g., "16123456"). + metadata is being requested (e.g., "16123456" or "18/915,708"). + The application number will be automatically sanitized. Returns: Optional[ApplicationMetaData]: An `ApplicationMetaData` object @@ -478,7 +524,7 @@ def get_application_metadata( is not available in the response. """ endpoint = self.ENDPOINTS["get_application_metadata"].format( - application_number=application_number + application_number=self.sanitize_application_number(application_number) ) response_data = self._make_request( method="GET", endpoint=endpoint, response_class=PatentDataResponse @@ -509,7 +555,7 @@ def get_application_adjustment( found or if PTA data is not available in the response. """ endpoint = self.ENDPOINTS["get_application_adjustment"].format( - application_number=application_number + application_number=self.sanitize_application_number(application_number) ) response_data = self._make_request( method="GET", endpoint=endpoint, response_class=PatentDataResponse @@ -541,7 +587,7 @@ def get_application_assignment( assignments. """ endpoint = self.ENDPOINTS["get_application_assignment"].format( - application_number=application_number + application_number=self.sanitize_application_number(application_number) ) response_data = self._make_request( method="GET", endpoint=endpoint, response_class=PatentDataResponse @@ -571,7 +617,7 @@ def get_application_attorney( be found or if no attorney data is available in the response. """ endpoint = self.ENDPOINTS["get_application_attorney"].format( - application_number=application_number + application_number=self.sanitize_application_number(application_number) ) response_data = self._make_request( method="GET", endpoint=endpoint, response_class=PatentDataResponse @@ -604,7 +650,7 @@ def get_application_continuity( links exist. """ endpoint = self.ENDPOINTS["get_application_continuity"].format( - application_number=application_number + application_number=self.sanitize_application_number(application_number) ) response_data = self._make_request( method="GET", endpoint=endpoint, response_class=PatentDataResponse @@ -636,7 +682,7 @@ def get_application_foreign_priority( is found but has no foreign priority claims. """ endpoint = self.ENDPOINTS["get_application_foreign_priority"].format( - application_number=application_number + application_number=self.sanitize_application_number(application_number) ) response_data = self._make_request( method="GET", endpoint=endpoint, response_class=PatentDataResponse @@ -668,7 +714,7 @@ def get_application_transactions( the application is found but has no recorded transaction events. """ endpoint = self.ENDPOINTS["get_application_transactions"].format( - application_number=application_number + application_number=self.sanitize_application_number(application_number) ) response_data = self._make_request( method="GET", endpoint=endpoint, response_class=PatentDataResponse @@ -713,7 +759,7 @@ def get_application_documents( is returned instead. """ endpoint = self.ENDPOINTS["get_application_documents"].format( - application_number=application_number + application_number=self.sanitize_application_number(application_number) ) params = {} @@ -758,7 +804,7 @@ def get_application_associated_documents( (e.g., PGPUB) does not exist for the application. """ endpoint = self.ENDPOINTS["get_application_associated_documents"].format( - application_number=application_number + application_number=self.sanitize_application_number(application_number) ) response_data = self._make_request( method="GET", endpoint=endpoint, response_class=PatentDataResponse @@ -987,22 +1033,25 @@ def download_archive( ) -> str: """Downloads Printed Metadata (XML data). These are XML files of the patent as printed. + Note: + See also `download_publication()` for a clearer method name with identical functionality. + Args: printed_metadata: ArchiveMetaData object containing download URL and metadata - file_name: Optional filename. If not provided, uses zip_file_name from metadata - destination_path: Optional directory path to save the archive + file_name: Optional filename. If not provided, uses xml_file_name from metadata + destination_path: Optional directory path to save the file overwrite: Whether to overwrite existing files. Default False Returns: - str: Path to the downloaded archive file + str: Path to the downloaded file Raises: - ValueError: If archive_metadata has no download URL + ValueError: If printed_metadata has no download URL FileExistsError: If file exists and overwrite=False """ # Validate we have a download URL if printed_metadata.file_location_uri is None: - raise ValueError("ArchiveMetaData must have a file_location_uri") + raise ValueError("PrintedMetaData must have a file_location_uri") # Get filename - either provided or from metadata if file_name is None: @@ -1036,3 +1085,70 @@ def download_archive( return self._download_file( url=printed_metadata.file_location_uri, file_path=final_file_path.as_posix() ) + + def download_publication( + self, + printed_metadata: PrintedMetaData, + file_name: Optional[str] = None, + destination_path: Optional[str] = None, + overwrite: bool = False, + ) -> str: + """Download a publication XML file (grant or pre-grant publication). + + This method downloads publication XML files from PrintedMetaData objects, + such as grant documents or pre-grant publications (pgpub). The filename + is automatically extracted from the metadata if not provided. + + Args: + printed_metadata: PrintedMetaData object containing the publication + download URL and filename information. Typically obtained from + `get_application_associated_documents()` or from PatentFileWrapper's + `grant_document_meta_data` or `pg_publication_document_meta_data`. + file_name: Optional custom filename. If not provided, uses the + `xml_file_name` from the metadata (e.g., "18915708_12307527.xml"). + destination_path: Optional directory path where the file should be saved. + If not provided, saves to the current directory. The directory will + be created if it doesn't exist. + overwrite: Whether to overwrite an existing file at the destination. + Default is False, which raises FileExistsError if file exists. + + Returns: + str: Absolute path to the downloaded publication file. + + Raises: + ValueError: If printed_metadata has no file_location_uri (download URL). + FileExistsError: If the file already exists and overwrite=False. + + Examples: + Download grant XML to a specific directory (auto-filename): + + >>> response = client.get_application_by_number("18/915,708") + >>> ifw = response + >>> grant_metadata = ifw.grant_document_meta_data + >>> path = client.download_publication(grant_metadata, destination_path="./downloads") + >>> print(path) + './downloads/18915708_12307527.xml' + + Download pgpub XML with custom filename: + + >>> pgpub_metadata = ifw.pg_publication_document_meta_data + >>> path = client.download_publication( + ... pgpub_metadata, + ... file_name="my_publication.xml", + ... destination_path="./downloads" + ... ) + >>> print(path) + './downloads/my_publication.xml' + + Download to current directory: + + >>> path = client.download_publication(grant_metadata) + >>> print(path) + './18915708_12307527.xml' + """ + return self.download_archive( + printed_metadata=printed_metadata, + file_name=file_name, + destination_path=destination_path, + overwrite=overwrite, + ) diff --git a/src/pyUSPTO/models/patent_data.py b/src/pyUSPTO/models/patent_data.py index 243600a..fe43a4c 100644 --- a/src/pyUSPTO/models/patent_data.py +++ b/src/pyUSPTO/models/patent_data.py @@ -274,6 +274,46 @@ def __len__(self) -> int: def __getitem__(self, index: int) -> Document: return self._documents[index] + def __str__(self) -> str: + """Returns a string representation showing document count and summary. + + Returns: + str: Human-readable summary of the DocumentBag. + """ + count = len(self._documents) + if count == 0: + return "DocumentBag(0 documents)" + + # Count unique document codes + doc_codes: Dict[str, int] = {} + for doc in self._documents: + code = doc.document_code or "Unknown" + doc_codes[code] = doc_codes.get(code, 0) + 1 + + # Format summary + if count == 1: + code = self._documents[0].document_code or "Unknown" + return f"DocumentBag(1 document: {code})" + + # Show top 3 most common document codes + sorted_codes = sorted(doc_codes.items(), key=lambda x: x[1], reverse=True) + top_codes = sorted_codes[:3] + code_summary = ", ".join(f"{code} ({cnt})" for code, cnt in top_codes) + + if len(sorted_codes) > 3: + remaining = len(sorted_codes) - 3 + return f"DocumentBag({count} documents: {code_summary}, +{remaining} more types)" + else: + return f"DocumentBag({count} documents: {code_summary})" + + def __repr__(self) -> str: + """Returns a detailed string representation for debugging. + + Returns: + str: Detailed representation of the DocumentBag. + """ + return f"DocumentBag(documents={self._documents!r})" + @classmethod def from_dict(cls, data: Dict[str, Any]) -> "DocumentBag": """Creates a `DocumentBag` instance from a dictionary representation. @@ -329,6 +369,9 @@ class Address: country_name: Full name of the country (e.g., "United States"). postal_address_category: Category of the address (e.g., "MAILING_ADDRESS"). correspondent_name_text: Name of the correspondent at this address. + country_or_state_code: Country or state code. + ict_state_code: International code for the state/region (USPTO format). + ict_country_code: International code for the country (USPTO format). """ name_line_one_text: Optional[str] = None @@ -345,6 +388,9 @@ class Address: country_name: Optional[str] = None postal_address_category: Optional[str] = None correspondent_name_text: Optional[str] = None + country_or_state_code: Optional[str] = None + ict_state_code: Optional[str] = None + ict_country_code: Optional[str] = None @classmethod def from_dict(cls, data: Dict[str, Any]) -> "Address": @@ -373,6 +419,9 @@ def from_dict(cls, data: Dict[str, Any]) -> "Address": country_name=data.get("countryName"), postal_address_category=data.get("postalAddressCategory"), correspondent_name_text=data.get("correspondentNameText"), + country_or_state_code=data.get("countryOrStateCode"), + ict_state_code=data.get("ictStateCode"), + ict_country_code=data.get("ictCountryCode"), ) def to_dict(self) -> Dict[str, Any]: @@ -397,6 +446,9 @@ def to_dict(self) -> Dict[str, Any]: "countryName": self.country_name, "postalAddressCategory": self.postal_address_category, "correspondentNameText": self.correspondent_name_text, + "countryOrStateCode": self.country_or_state_code, + "ictStateCode": self.ict_state_code, + "ictCountryCode": self.ict_country_code, } @@ -965,7 +1017,7 @@ class Assignment: """Represents a patent assignment, detailing the transfer of rights. Includes information about the reel and frame, document location, dates, conveyance text, - and bags of assignors, assignees, and correspondence addresses. + and bags of assignors, assignees, correspondence address, and domestic representative. Attributes: reel_number: Reel number for the assignment record. @@ -977,9 +1029,12 @@ class Assignment: assignment_recorded_date: Date the assignment was recorded by USPTO. assignment_mailed_date: Date the assignment notification was mailed. conveyance_text: Text describing the nature of the conveyance. + image_available_status_code: Code to indicate the availability of the image. + attorney_docket_number: Attorney docket number for the assignment. assignor_bag: List of `Assignor` objects. assignee_bag: List of `Assignee` objects. - correspondence_address_bag: List of `Address` objects for correspondence. + correspondence_address: `Address` object for correspondence (single object). + domestic_representative: `Address` object for the domestic representative. """ reel_number: Optional[int] = None @@ -991,9 +1046,12 @@ class Assignment: assignment_recorded_date: Optional[date] = None assignment_mailed_date: Optional[date] = None conveyance_text: Optional[str] = None + image_available_status_code: Optional[bool] = None + attorney_docket_number: Optional[str] = None assignor_bag: List[Assignor] = field(default_factory=list) assignee_bag: List[Assignee] = field(default_factory=list) - correspondence_address_bag: List[Address] = field(default_factory=list) + correspondence_address: Optional[Address] = None + domestic_representative: Optional[Address] = None @classmethod def from_dict(cls, data: Dict[str, Any]) -> "Assignment": @@ -1015,11 +1073,21 @@ def from_dict(cls, data: Dict[str, Any]) -> "Assignment": for a in data.get("assigneeBag", []) if isinstance(a, dict) ] - addrs = [ - Address.from_dict(a) - for a in data.get("correspondenceAddressBag", []) - if isinstance(a, dict) - ] + + # Parse correspondence address (single object, not bag) + corr_addr_data = data.get("correspondenceAddress") + corr_addr = ( + Address.from_dict(corr_addr_data) + if isinstance(corr_addr_data, dict) + else None + ) + + # Parse domestic representative + dom_rep_data = data.get("domesticRepresentative") + dom_rep = ( + Address.from_dict(dom_rep_data) if isinstance(dom_rep_data, dict) else None + ) + return cls( reel_number=data.get("reelNumber"), frame_number=data.get("frameNumber"), @@ -1030,9 +1098,12 @@ def from_dict(cls, data: Dict[str, Any]) -> "Assignment": assignment_recorded_date=parse_to_date(data.get("assignmentRecordedDate")), assignment_mailed_date=parse_to_date(data.get("assignmentMailedDate")), conveyance_text=data.get("conveyanceText"), + image_available_status_code=data.get("imageAvailableStatusCode"), + attorney_docket_number=data.get("attorneyDocketNumber"), assignor_bag=assignors, assignee_bag=assignees, - correspondence_address_bag=addrs, + correspondence_address=corr_addr, + domestic_representative=dom_rep, ) def to_dict(self) -> Dict[str, Any]: @@ -1051,11 +1122,20 @@ def to_dict(self) -> Dict[str, Any]: "assignmentRecordedDate": serialize_date(self.assignment_recorded_date), "assignmentMailedDate": serialize_date(self.assignment_mailed_date), "conveyanceText": self.conveyance_text, + "imageAvailableStatusCode": self.image_available_status_code, + "attorneyDocketNumber": self.attorney_docket_number, "assignorBag": [a.to_dict() for a in self.assignor_bag], "assigneeBag": [a.to_dict() for a in self.assignee_bag], - "correspondenceAddressBag": [ - a.to_dict() for a in self.correspondence_address_bag - ], + "correspondenceAddress": ( + self.correspondence_address.to_dict() + if self.correspondence_address + else None + ), + "domesticRepresentative": ( + self.domestic_representative.to_dict() + if self.domestic_representative + else None + ), } @@ -1322,7 +1402,6 @@ class PatentTermAdjustmentHistoryData: applicant_day_delay_quantity: Number of days of delay attributable to the applicant for this event. event_description_text: Textual description of the PTA event. event_sequence_number: Sequence number of this event in the PTA history. - ip_office_day_delay_quantity: Number of days of delay attributable to the IP office for this event. originating_event_sequence_number: Sequence number of an event that originated this event. pta_pte_code: Code indicating if the event relates to PTA or Patent Term Extension (PTE). """ @@ -1331,7 +1410,6 @@ class PatentTermAdjustmentHistoryData: applicant_day_delay_quantity: Optional[float] = None event_description_text: Optional[str] = None event_sequence_number: Optional[float] = None - ip_office_day_delay_quantity: Optional[float] = None originating_event_sequence_number: Optional[float] = None pta_pte_code: Optional[str] = None @@ -1350,7 +1428,6 @@ def from_dict(cls, data: Dict[str, Any]) -> "PatentTermAdjustmentHistoryData": applicant_day_delay_quantity=data.get("applicantDayDelayQuantity"), event_description_text=data.get("eventDescriptionText"), event_sequence_number=data.get("eventSequenceNumber"), - ip_office_day_delay_quantity=data.get("ipOfficeDayDelayQuantity"), originating_event_sequence_number=data.get( "originatingEventSequenceNumber" ), @@ -1374,8 +1451,6 @@ def to_dict(self) -> Dict[str, Any]: final_dict["eventDescriptionText"] = self.event_description_text if self.event_sequence_number is not None: final_dict["eventSequenceNumber"] = self.event_sequence_number - if self.ip_office_day_delay_quantity is not None: - final_dict["ipOfficeDayDelayQuantity"] = self.ip_office_day_delay_quantity if self.originating_event_sequence_number is not None: final_dict["originatingEventSequenceNumber"] = ( self.originating_event_sequence_number @@ -1398,8 +1473,6 @@ class PatentTermAdjustmentData: applicant_day_delay_quantity: Total days of delay attributable to the applicant. b_delay_quantity: Number of days of 'B' delay. c_delay_quantity: Number of days of 'C' delay. - filing_date: The filing date of the application. - grant_date: The grant date of the patent. non_overlapping_day_quantity: Number of non-overlapping delay days. overlapping_day_quantity: Number of overlapping delay days. ip_office_day_delay_quantity: Total days of delay attributable to the IP office. @@ -1411,8 +1484,6 @@ class PatentTermAdjustmentData: applicant_day_delay_quantity: Optional[float] = None b_delay_quantity: Optional[float] = None c_delay_quantity: Optional[float] = None - filing_date: Optional[date] = None - grant_date: Optional[date] = None non_overlapping_day_quantity: Optional[float] = None overlapping_day_quantity: Optional[float] = None ip_office_day_delay_quantity: Optional[float] = None @@ -1441,8 +1512,6 @@ def from_dict(cls, data: Dict[str, Any]) -> "PatentTermAdjustmentData": applicant_day_delay_quantity=data.get("applicantDayDelayQuantity"), b_delay_quantity=data.get("bDelayQuantity"), c_delay_quantity=data.get("cDelayQuantity"), - filing_date=parse_to_date(data.get("filingDate")), - grant_date=parse_to_date(data.get("grantDate")), non_overlapping_day_quantity=data.get("nonOverlappingDayQuantity"), overlapping_day_quantity=data.get("overlappingDayQuantity"), ip_office_day_delay_quantity=data.get("ipOfficeDayDelayQuantity"), @@ -1458,8 +1527,6 @@ def to_dict(self) -> Dict[str, Any]: Dict[str, Any]: Dictionary representation. """ d = asdict(self) - d["filingDate"] = serialize_date(self.filing_date) - d["grantDate"] = serialize_date(self.grant_date) d["patentTermAdjustmentHistoryDataBag"] = [ h.to_dict() for h in self.patent_term_adjustment_history_data_bag ] @@ -1672,11 +1739,14 @@ def is_pre_aia(self) -> Optional[bool]: return not self.first_inventor_to_file_indicator @classmethod - def from_dict(cls, data: Dict[str, Any]) -> "ApplicationMetaData": + def from_dict( + cls, data: Dict[str, Any], include_raw_data: bool = False + ) -> "ApplicationMetaData": """Creates an `ApplicationMetaData` instance from a dictionary. Args: data (Dict[str, Any]): Dictionary with application metadata. + include_raw_data (bool): If True, store the raw JSON for debugging. Returns: ApplicationMetaData: An instance of `ApplicationMetaData`. @@ -1751,7 +1821,7 @@ def from_dict(cls, data: Dict[str, Any]) -> "ApplicationMetaData": cpc_classification_bag=data.get("cpcClassificationBag", []), applicant_bag=app_bag, inventor_bag=inv_bag, - raw_data=json.dumps(data), + raw_data=json.dumps(data) if include_raw_data else None, ) def to_dict(self) -> Dict[str, Any]: @@ -1869,18 +1939,21 @@ class PatentFileWrapper: last_ingestion_date_time: Optional[datetime] = None @classmethod - def from_dict(cls, data: Dict[str, Any]) -> "PatentFileWrapper": + def from_dict( + cls, data: Dict[str, Any], include_raw_data: bool = False + ) -> "PatentFileWrapper": """Creates a `PatentFileWrapper` instance from a dictionary. Args: data (Dict[str, Any]): Dictionary with patent file wrapper data. + include_raw_data (bool): If True, store the raw JSON for debugging. Returns: PatentFileWrapper: An instance of `PatentFileWrapper`. """ amd_json = data.get("applicationMetaData") amd = ( - ApplicationMetaData.from_dict(amd_json) + ApplicationMetaData.from_dict(amd_json, include_raw_data=include_raw_data) if isinstance(amd_json, dict) else None ) @@ -2039,7 +2112,7 @@ def from_dict( PatentDataResponse: An instance of `PatentDataResponse`. """ wrappers = [ - PatentFileWrapper.from_dict(w) + PatentFileWrapper.from_dict(w, include_raw_data=include_raw_data) for w in data.get("patentFileWrapperDataBag", []) if isinstance(w, dict) ] diff --git a/tests/clients/test_base.py b/tests/clients/test_base.py index 014c9e0..9c86c79 100644 --- a/tests/clients/test_base.py +++ b/tests/clients/test_base.py @@ -718,6 +718,67 @@ def test_extract_filename_complex(self) -> None: assert filename == "report.pdf" +class TestMimeTypeMapping: + """Tests for _get_extension_from_mime_type method.""" + + def test_mime_type_pdf(self) -> None: + """Test mapping application/pdf to .pdf extension.""" + ext = BaseUSPTOClient._get_extension_from_mime_type("application/pdf") + assert ext == ".pdf" + + def test_mime_type_tiff(self) -> None: + """Test mapping image/tiff to .tif extension.""" + ext = BaseUSPTOClient._get_extension_from_mime_type("image/tiff") + assert ext == ".tif" + + def test_mime_type_tif_variant(self) -> None: + """Test mapping image/tif to .tif extension.""" + ext = BaseUSPTOClient._get_extension_from_mime_type("image/tif") + assert ext == ".tif" + + def test_mime_type_xml_application(self) -> None: + """Test mapping application/xml to .xml extension.""" + ext = BaseUSPTOClient._get_extension_from_mime_type("application/xml") + assert ext == ".xml" + + def test_mime_type_xml_text(self) -> None: + """Test mapping text/xml to .xml extension.""" + ext = BaseUSPTOClient._get_extension_from_mime_type("text/xml") + assert ext == ".xml" + + def test_mime_type_zip(self) -> None: + """Test mapping application/zip to .zip extension.""" + ext = BaseUSPTOClient._get_extension_from_mime_type("application/zip") + assert ext == ".zip" + + def test_mime_type_with_charset(self) -> None: + """Test MIME type with charset parameter.""" + ext = BaseUSPTOClient._get_extension_from_mime_type( + "application/pdf; charset=utf-8" + ) + assert ext == ".pdf" + + def test_mime_type_case_insensitive(self) -> None: + """Test MIME type mapping is case-insensitive.""" + ext = BaseUSPTOClient._get_extension_from_mime_type("APPLICATION/PDF") + assert ext == ".pdf" + + def test_mime_type_unmapped(self) -> None: + """Test unmapped MIME type returns None.""" + ext = BaseUSPTOClient._get_extension_from_mime_type("application/unknown") + assert ext is None + + def test_mime_type_empty(self) -> None: + """Test empty MIME type returns None.""" + ext = BaseUSPTOClient._get_extension_from_mime_type("") + assert ext is None + + def test_mime_type_none(self) -> None: + """Test None MIME type returns None.""" + ext = BaseUSPTOClient._get_extension_from_mime_type(None) + assert ext is None + + class TestSaveResponseToFile: """Tests for _save_response_to_file method.""" @@ -768,3 +829,143 @@ def test_save_to_directory_without_content_disposition( match="file_path is a directory .* but Content-Disposition header does not contain a filename", ): client._save_response_to_file(mock_response, str(tmp_path)) + + @patch("builtins.open", new_callable=mock_open) + def test_save_without_extension_uses_content_type_pdf( + self, mock_file_open: MagicMock, tmp_path: Any + ) -> None: + """Test saving file without extension adds extension from Content-Type (PDF).""" + client: BaseUSPTOClient[Any] = BaseUSPTOClient( + api_key="test", base_url="https://test.com" + ) + + # Mock response with Content-Type but no Content-Disposition + mock_response = MagicMock() + mock_response.headers = {"Content-Type": "application/pdf"} + mock_response.iter_content.return_value = [b"pdf data"] + + # Save to file without extension + file_path = tmp_path / "document" + result = client._save_response_to_file(mock_response, str(file_path)) + + # Verify extension was added + expected_path = tmp_path / "document.pdf" + mock_file_open.assert_called_once_with(file=str(expected_path), mode="wb") + assert result == str(expected_path) + + @patch("builtins.open", new_callable=mock_open) + def test_save_without_extension_uses_content_type_tiff( + self, mock_file_open: MagicMock, tmp_path: Any + ) -> None: + """Test saving file without extension adds extension from Content-Type (TIFF).""" + client: BaseUSPTOClient[Any] = BaseUSPTOClient( + api_key="test", base_url="https://test.com" + ) + + # Mock response with TIFF Content-Type + mock_response = MagicMock() + mock_response.headers = {"Content-Type": "image/tiff"} + mock_response.iter_content.return_value = [b"tiff data"] + + # Save to file without extension + file_path = tmp_path / "image" + result = client._save_response_to_file(mock_response, str(file_path)) + + # Verify .tif extension was added + expected_path = tmp_path / "image.tif" + mock_file_open.assert_called_once_with(file=str(expected_path), mode="wb") + assert result == str(expected_path) + + @patch("builtins.open", new_callable=mock_open) + def test_save_with_existing_extension_ignores_content_type( + self, mock_file_open: MagicMock, tmp_path: Any + ) -> None: + """Test saving file with existing extension ignores Content-Type.""" + client: BaseUSPTOClient[Any] = BaseUSPTOClient( + api_key="test", base_url="https://test.com" + ) + + # Mock response with Content-Type + mock_response = MagicMock() + mock_response.headers = {"Content-Type": "application/pdf"} + mock_response.iter_content.return_value = [b"data"] + + # Save to file with existing extension + file_path = tmp_path / "document.txt" + result = client._save_response_to_file(mock_response, str(file_path)) + + # Verify original extension was kept + expected_path = tmp_path / "document.txt" + mock_file_open.assert_called_once_with(file=str(expected_path), mode="wb") + assert result == str(expected_path) + + @patch("builtins.open", new_callable=mock_open) + def test_save_without_extension_unmapped_mime_type( + self, mock_file_open: MagicMock, tmp_path: Any + ) -> None: + """Test saving file with unmapped MIME type saves without extension.""" + client: BaseUSPTOClient[Any] = BaseUSPTOClient( + api_key="test", base_url="https://test.com" + ) + + # Mock response with unmapped Content-Type + mock_response = MagicMock() + mock_response.headers = {"Content-Type": "application/unknown"} + mock_response.iter_content.return_value = [b"data"] + + # Save to file without extension + file_path = tmp_path / "document" + result = client._save_response_to_file(mock_response, str(file_path)) + + # Verify no extension was added + expected_path = tmp_path / "document" + mock_file_open.assert_called_once_with(file=str(expected_path), mode="wb") + assert result == str(expected_path) + + @patch("builtins.open", new_callable=mock_open) + def test_save_without_extension_no_content_type( + self, mock_file_open: MagicMock, tmp_path: Any + ) -> None: + """Test saving file without Content-Type header saves without extension.""" + client: BaseUSPTOClient[Any] = BaseUSPTOClient( + api_key="test", base_url="https://test.com" + ) + + # Mock response without Content-Type header + mock_response = MagicMock() + mock_response.headers = {} + mock_response.iter_content.return_value = [b"data"] + + # Save to file without extension + file_path = tmp_path / "document" + result = client._save_response_to_file(mock_response, str(file_path)) + + # Verify no extension was added + expected_path = tmp_path / "document" + mock_file_open.assert_called_once_with(file=str(expected_path), mode="wb") + assert result == str(expected_path) + + @patch("builtins.open", new_callable=mock_open) + def test_save_content_disposition_takes_precedence_over_content_type( + self, mock_file_open: MagicMock, tmp_path: Any + ) -> None: + """Test Content-Disposition filename takes precedence over Content-Type extension.""" + client: BaseUSPTOClient[Any] = BaseUSPTOClient( + api_key="test", base_url="https://test.com" + ) + + # Mock response with both Content-Disposition and Content-Type + mock_response = MagicMock() + mock_response.headers = { + "Content-Disposition": 'attachment; filename="report.xml"', + "Content-Type": "application/pdf", # Different type + } + mock_response.iter_content.return_value = [b"data"] + + # Save to directory (will use Content-Disposition) + result = client._save_response_to_file(mock_response, str(tmp_path)) + + # Verify Content-Disposition filename was used (not Content-Type) + expected_path = tmp_path / "report.xml" + mock_file_open.assert_called_once_with(file=str(expected_path), mode="wb") + assert result == str(expected_path) diff --git a/tests/clients/test_patent_data_clients.py b/tests/clients/test_patent_data_clients.py index 7470a2f..737b57e 100644 --- a/tests/clients/test_patent_data_clients.py +++ b/tests/clients/test_patent_data_clients.py @@ -631,7 +631,7 @@ def test_get_application_by_number_empty_bag_returns_none( """Test get_application_by_number returns None if patentFileWrapperDataBag is empty.""" client, mock_make_request = client_with_mocked_request mock_make_request.return_value = mock_patent_data_response_empty - app_num_to_request = "nonexistent123" + app_num_to_request = "00000000" result = client.get_application_by_number(application_number=app_num_to_request) assert result is None @@ -682,7 +682,7 @@ def test_get_application_documents( ) -> None: """Test retrieval of application documents.""" client, mock_make_request = client_with_mocked_request - app_num = "appDoc123" + app_num = "12345678" mock_response_dict = { "documentBag": [ { @@ -712,7 +712,7 @@ def test_get_application_documents_with_document_code_filter( ) -> None: """Test retrieval of application documents filtered by document codes.""" client, mock_make_request = client_with_mocked_request - app_num = "appDoc456" + app_num = "12345678" mock_response_dict = { "documentBag": [ { @@ -744,7 +744,7 @@ def test_get_application_documents_with_date_filter( ) -> None: """Test retrieval of application documents filtered by official date range.""" client, mock_make_request = client_with_mocked_request - app_num = "appDoc789" + app_num = "12345678" mock_response_dict = { "documentBag": [ { @@ -777,7 +777,7 @@ def test_get_application_documents_with_combined_filters( ) -> None: """Test retrieval of application documents with multiple filters combined.""" client, mock_make_request = client_with_mocked_request - app_num = "appDoc999" + app_num = "12345678" mock_response_dict = {"documentBag": []} mock_make_request.return_value = mock_response_dict result = client.get_application_documents( @@ -804,7 +804,7 @@ def test_get_application_documents_with_partial_date_filter( ) -> None: """Test retrieval with only one date boundary specified.""" client, mock_make_request = client_with_mocked_request - app_num = "appDoc111" + app_num = "12345678" mock_response_dict = {"documentBag": []} mock_make_request.return_value = mock_response_dict @@ -1095,6 +1095,7 @@ def test_download_file_success( # Verify file operations - use str(Path()) to normalize path for platform from pathlib import Path + expected_path = str(Path(file_path)) mock_file_open.assert_called_once_with(file=expected_path, mode="wb") mock_file_open().write.assert_has_calls( @@ -1131,6 +1132,7 @@ def test_download_file_filters_empty_chunks( ) -> None: """Test that empty chunks are filtered out.""" mock_response = MagicMock(spec=requests.Response) + mock_response.headers = {} # Add headers to mock mock_response.iter_content.return_value = [b"data", b"", None, b"more"] mock_make_request.return_value = mock_response @@ -1247,11 +1249,141 @@ def test_get_ifw_by_pct_app_number( # Should call get_application_by_number mock_make_request.assert_called_once_with( method="GET", - endpoint=f"api/v1/patent/applications/{pct_app}", + endpoint=f"api/v1/patent/applications/PCTUS24012345", response_class=PatentDataResponse, ) assert result is mock_patent_file_wrapper + def test_get_ifw_by_short_pct_app_number( + self, + client_with_mocked_request: tuple[PatentDataClient, MagicMock], + mock_patent_file_wrapper: PatentFileWrapper, + ) -> None: + """Test PCT application number sanitization with 2-digit year format (US24 vs US2024). + + Verifies that PCT numbers with short year format (PCT/US24/012345) are correctly + sanitized to PCTUS24012345 before making API request. + + Note: This will trigger a data mismatch warning because the mock_patent_file_wrapper + has application_number_text='12345678' but we're requesting a PCT number. + This is expected test behavior for validating the warning system. + """ + client, mock_make_request = client_with_mocked_request + mock_make_request.return_value = PatentDataResponse( + count=1, patent_file_wrapper_data_bag=[mock_patent_file_wrapper] + ) + + pct_app = "PCT/US24/012345" + + # The mismatch between PCT number and regular app number triggers warning + with pytest.warns(USPTODataMismatchWarning): + result = client.get_IFW_metadata(PCT_app_number=pct_app) + + # Should call get_application_by_number + mock_make_request.assert_called_once_with( + method="GET", + endpoint=f"api/v1/patent/applications/PCTUS24012345", + response_class=PatentDataResponse, + ) + assert result is mock_patent_file_wrapper + + def test_get_ifw_by_pct_app_number_malformed( + self, + client_with_mocked_request: tuple[PatentDataClient, MagicMock], + mock_patent_file_wrapper: PatentFileWrapper, + ) -> None: + """Test PCT application number validation rejects malformed format missing first slash. + + Verifies that PCT numbers missing the first slash (PCTUS2024/012345 instead of + PCT/US2024/012345) raise ValueError with descriptive error message. + """ + client, mock_make_request = client_with_mocked_request + mock_make_request.return_value = PatentDataResponse( + count=1, patent_file_wrapper_data_bag=[mock_patent_file_wrapper] + ) + + pct_app = "PCTUS2024/012345" + + # The malformed PCT number triggers error + with pytest.raises( + ValueError, + match="Invalid PCT application format: PCTUS2024/012345. Expected PCT/CCYYYY/NNNNNN", + ): + client.get_IFW_metadata(PCT_app_number=pct_app) + + def test_get_ifw_by_pct_app_year_corrupted( + self, + client_with_mocked_request: tuple[PatentDataClient, MagicMock], + mock_patent_file_wrapper: PatentFileWrapper, + ) -> None: + """Test PCT application number validation rejects invalid year length. + + Verifies that PCT numbers with incorrect year length (PCT/US224/012345 with 3-digit + year instead of 2 or 4 digits) raise ValueError with descriptive error message. + """ + client, mock_make_request = client_with_mocked_request + mock_make_request.return_value = PatentDataResponse( + count=1, patent_file_wrapper_data_bag=[mock_patent_file_wrapper] + ) + + pct_app = "PCT/US224/012345" + + # The malformed PCT number triggers error + with pytest.raises( + ValueError, + match="Invalid PCT year length in: US224. Expected CCYYYY or CCYY.", + ): + client.get_IFW_metadata(PCT_app_number=pct_app) + + def test_get_ifw_by_pct_app_year_malformed( + self, + client_with_mocked_request: tuple[PatentDataClient, MagicMock], + mock_patent_file_wrapper: PatentFileWrapper, + ) -> None: + """Test PCT application number validation rejects non-numeric year. + + Verifies that PCT numbers with non-digit characters in year field + (PCT/USA2024/012345 instead of PCT/US2024/012345) raise ValueError with + descriptive error message. + """ + client, mock_make_request = client_with_mocked_request + mock_make_request.return_value = PatentDataResponse( + count=1, patent_file_wrapper_data_bag=[mock_patent_file_wrapper] + ) + + pct_app = "PCT/USA2024/012345" + + # The malformed PCT number triggers error + with pytest.raises( + ValueError, + match="Invalid PCT year in: USA2024. Must be digits.", + ): + client.get_IFW_metadata(PCT_app_number=pct_app) + + def test_get_ifw_by_pct_app_serial_malformed( + self, + client_with_mocked_request: tuple[PatentDataClient, MagicMock], + mock_patent_file_wrapper: PatentFileWrapper, + ) -> None: + """Test PCT application number validation rejects non-numeric serial number. + + Verifies that PCT numbers with non-numeric serial number (PCT/US2024/A12345 + instead of PCT/US2024/012345) raise ValueError with descriptive error message. + """ + client, mock_make_request = client_with_mocked_request + mock_make_request.return_value = PatentDataResponse( + count=1, patent_file_wrapper_data_bag=[mock_patent_file_wrapper] + ) + + pct_app = "PCT/US2024/A12345" + + # The malformed PCT number triggers error + with pytest.raises( + ValueError, + match="Invalid PCT serial: A12345. Must be numeric.", + ): + client.get_IFW_metadata(PCT_app_number=pct_app) + def test_get_ifw_by_pct_pub_number( self, client_with_mocked_request: tuple[PatentDataClient, MagicMock], @@ -1431,7 +1563,7 @@ def test_download_archive_missing_url( ) with pytest.raises( - ValueError, match="ArchiveMetaData must have a file_location_uri" + ValueError, match="PrintedMetaData must have a file_location_uri" ): client.download_archive(printed_metadata=metadata_no_url) @@ -1528,6 +1660,137 @@ def test_download_archive_last_resort_filename( ) assert result == expected_path + # Tests for download_publication() - delegates to download_archive() + @patch("pathlib.Path.exists") + @patch("pathlib.Path.mkdir") + def test_download_publication_basic( + self, + mock_mkdir: MagicMock, + mock_exists: MagicMock, + client_with_mocked_download: tuple[PatentDataClient, MagicMock], + sample_printed_metadata: PrintedMetaData, + ) -> None: + """Test basic publication download.""" + client, mock_download_file = client_with_mocked_download + mock_exists.return_value = False + + expected_path = "/downloads/patent_12345.xml" + mock_download_file.return_value = expected_path + + result = client.download_publication( + printed_metadata=sample_printed_metadata, destination_path="/downloads" + ) + + mock_download_file.assert_called_once_with( + url=sample_printed_metadata.file_location_uri, file_path=expected_path + ) + assert result == expected_path + + @patch("pathlib.Path.exists") + @patch("pathlib.Path.mkdir") + def test_download_publication_custom_filename( + self, + mock_mkdir: MagicMock, + mock_exists: MagicMock, + client_with_mocked_download: tuple[PatentDataClient, MagicMock], + sample_printed_metadata: PrintedMetaData, + ) -> None: + """Test publication download with custom filename.""" + client, mock_download_file = client_with_mocked_download + mock_exists.return_value = False + + custom_name = "my_grant.xml" + expected_path = "/downloads/my_grant.xml" + mock_download_file.return_value = expected_path + + result = client.download_publication( + printed_metadata=sample_printed_metadata, + file_name=custom_name, + destination_path="/downloads", + ) + + mock_download_file.assert_called_once_with( + url=sample_printed_metadata.file_location_uri, file_path=expected_path + ) + assert result == expected_path + + @patch("pathlib.Path.exists") + def test_download_publication_no_destination_path( + self, + mock_exists: MagicMock, + client_with_mocked_download: tuple[PatentDataClient, MagicMock], + sample_printed_metadata: PrintedMetaData, + ) -> None: + """Test publication download with no destination path (current directory).""" + client, mock_download_file = client_with_mocked_download + mock_exists.return_value = False + + expected_path = "patent_12345.xml" + mock_download_file.return_value = expected_path + + result = client.download_publication(printed_metadata=sample_printed_metadata) + + mock_download_file.assert_called_once_with( + url=sample_printed_metadata.file_location_uri, file_path=expected_path + ) + assert result == expected_path + + def test_download_publication_missing_url( + self, client_with_mocked_download: tuple[PatentDataClient, MagicMock] + ) -> None: + """Test download_publication raises ValueError when no download URL.""" + client, mock_download_file = client_with_mocked_download + + metadata_no_url = PrintedMetaData( + xml_file_name="test.xml", file_location_uri=None + ) + + with pytest.raises( + ValueError, match="PrintedMetaData must have a file_location_uri" + ): + client.download_publication(printed_metadata=metadata_no_url) + + mock_download_file.assert_not_called() + + @patch("pathlib.Path.exists") + def test_download_publication_file_exists_no_overwrite( + self, + mock_exists: MagicMock, + client_with_mocked_download: tuple[PatentDataClient, MagicMock], + sample_printed_metadata: PrintedMetaData, + ) -> None: + """Test download_publication raises FileExistsError when file exists.""" + client, mock_download_file = client_with_mocked_download + mock_exists.return_value = True + + with pytest.raises( + FileExistsError, match="File already exists.*Use overwrite=True" + ): + client.download_publication(printed_metadata=sample_printed_metadata) + + mock_download_file.assert_not_called() + + @patch("pathlib.Path.exists") + def test_download_publication_overwrite_existing( + self, + mock_exists: MagicMock, + client_with_mocked_download: tuple[PatentDataClient, MagicMock], + sample_printed_metadata: PrintedMetaData, + ) -> None: + """Test download_publication overwrites when overwrite=True.""" + client, mock_download_file = client_with_mocked_download + mock_exists.return_value = True + + expected_path = "patent_12345.xml" + mock_download_file.return_value = expected_path + + result = client.download_publication( + printed_metadata=sample_printed_metadata, overwrite=True + ) + + mock_download_file.assert_called_once() + assert result == expected_path + class TestPatentApplicationDataRetrieval: """Tests for data retrieval of patent application search results using get_search_results.""" @@ -2024,7 +2287,7 @@ def test_get_application_by_number_app_num_mismatch_in_bag( the data inconsistency. """ client, mock_make_request = client_with_mocked_request - requested_app_num = "DIFFERENT_APP_NUM_999" + requested_app_num = "87654321" response_with_original_wrapper = PatentDataResponse( count=1, patent_file_wrapper_data_bag=[mock_patent_file_wrapper] ) @@ -2032,7 +2295,7 @@ def test_get_application_by_number_app_num_mismatch_in_bag( with pytest.warns( USPTODataMismatchWarning, - match="API returned application number '12345678' but requested 'DIFFERENT_APP_NUM_999'", + match="API returned application number '12345678' but requested '87654321'", ): result = client.get_application_by_number( application_number=requested_app_num @@ -2049,7 +2312,7 @@ def test_get_application_by_number_unexpected_response_type( mock_make_request.return_value = ["not", "a", "PatentDataResponse"] with pytest.raises(AssertionError): - client.get_application_by_number(application_number="123") + client.get_application_by_number(application_number="32165487") def test_api_error_handling( self, client_with_mocked_request: tuple[PatentDataClient, MagicMock] @@ -2148,15 +2411,11 @@ def test_sanitize_invalid_series_code_format_raises( ) -> None: """Test invalid series code format raises ValueError.""" # Wrong series length - with pytest.raises( - ValueError, match="Expected series code format: NN/NNNNNN" - ): + with pytest.raises(ValueError, match="Expected series code format: NN/NNNNNN"): patent_data_client.sanitize_application_number("8/123456") # 1 digit series # Wrong serial length - with pytest.raises( - ValueError, match="Expected series code format: NN/NNNNNN" - ): + with pytest.raises(ValueError, match="Expected series code format: NN/NNNNNN"): patent_data_client.sanitize_application_number("08/12345") # 5 digit serial # Non-numeric series @@ -2168,9 +2427,7 @@ def test_sanitize_invalid_series_code_format_raises( patent_data_client.sanitize_application_number("08/ABC456") # Multiple slashes - with pytest.raises( - ValueError, match="Expected format: NNNNNNNN or NN/NNNNNN" - ): + with pytest.raises(ValueError, match="Expected format: NNNNNNNN or NN/NNNNNN"): patent_data_client.sanitize_application_number("08/123/456") diff --git a/tests/models/test_patent_data_models.py b/tests/models/test_patent_data_models.py index 3c3572b..439078a 100644 --- a/tests/models/test_patent_data_models.py +++ b/tests/models/test_patent_data_models.py @@ -82,6 +82,9 @@ def sample_address_data() -> Dict[str, Any]: "countryName": "United States", "postalAddressCategory": "Mailing", "correspondentNameText": "Test Correspondent", + "countryOrStateCode": None, + "ictStateCode": None, + "ictCountryCode": None, } @@ -748,6 +751,94 @@ def test_document_bag_iterable(self) -> None: count += 1 assert count == 1 + def test_document_bag_str_empty(self) -> None: + """Test __str__ with empty DocumentBag.""" + doc_bag = DocumentBag(documents=[]) + assert str(doc_bag) == "DocumentBag(0 documents)" + + def test_document_bag_str_single_document(self) -> None: + """Test __str__ with single document.""" + doc = Document(document_identifier="doc1", document_code="OATH") + doc_bag = DocumentBag(documents=[doc]) + assert str(doc_bag) == "DocumentBag(1 document: OATH)" + + def test_document_bag_str_single_document_no_code(self) -> None: + """Test __str__ with single document without document_code.""" + doc = Document(document_identifier="doc1", document_code=None) + doc_bag = DocumentBag(documents=[doc]) + assert str(doc_bag) == "DocumentBag(1 document: Unknown)" + + def test_document_bag_str_multiple_documents_same_code(self) -> None: + """Test __str__ with multiple documents of same type.""" + docs = [ + Document(document_identifier=f"doc{i}", document_code="OATH") + for i in range(5) + ] + doc_bag = DocumentBag(documents=docs) + assert str(doc_bag) == "DocumentBag(5 documents: OATH (5))" + + def test_document_bag_str_multiple_document_types(self) -> None: + """Test __str__ with multiple document types (<=3 types).""" + docs = [ + Document(document_identifier="doc1", document_code="OATH"), + Document(document_identifier="doc2", document_code="OATH"), + Document(document_identifier="doc3", document_code="CLM"), + Document(document_identifier="doc4", document_code="CLM"), + Document(document_identifier="doc5", document_code="SPEC"), + ] + doc_bag = DocumentBag(documents=docs) + result = str(doc_bag) + assert result.startswith("DocumentBag(5 documents:") + assert "OATH (2)" in result + assert "CLM (2)" in result + assert "SPEC (1)" in result + assert "+0 more" not in result # Only 3 types, no "more" + + def test_document_bag_str_many_document_types(self) -> None: + """Test __str__ with more than 3 document types.""" + docs = [ + Document(document_identifier="doc1", document_code="OATH"), + Document(document_identifier="doc2", document_code="OATH"), + Document(document_identifier="doc3", document_code="OATH"), + Document(document_identifier="doc4", document_code="CLM"), + Document(document_identifier="doc5", document_code="CLM"), + Document(document_identifier="doc6", document_code="SPEC"), + Document(document_identifier="doc7", document_code="DRFT"), + Document(document_identifier="doc8", document_code="IDS"), + ] + doc_bag = DocumentBag(documents=docs) + result = str(doc_bag) + assert result.startswith("DocumentBag(8 documents:") + # Should show top 3 most common + assert "OATH (3)" in result + assert "CLM (2)" in result + assert "+2 more types" in result # 5 total types - 3 shown = 2 more + + def test_document_bag_str_mixed_codes_and_none(self) -> None: + """Test __str__ with mix of document codes and None.""" + docs = [ + Document(document_identifier="doc1", document_code="OATH"), + Document(document_identifier="doc2", document_code=None), + Document(document_identifier="doc3", document_code="CLM"), + Document(document_identifier="doc4", document_code=None), + ] + doc_bag = DocumentBag(documents=docs) + result = str(doc_bag) + assert result.startswith("DocumentBag(4 documents:") + assert "Unknown (2)" in result + assert "OATH (1)" in result or "CLM (1)" in result + + def test_document_bag_repr(self) -> None: + """Test __repr__ returns detailed representation.""" + doc1 = Document(document_identifier="doc1", document_code="OATH") + doc2 = Document(document_identifier="doc2", document_code="CLM") + doc_bag = DocumentBag(documents=[doc1, doc2]) + result = repr(doc_bag) + assert result.startswith("DocumentBag(documents=(") + assert "Document(" in result + assert "doc1" in result + assert "doc2" in result + class TestAddress: """Tests for the Address class.""" @@ -793,6 +884,9 @@ def test_address_to_dict_empty(self) -> None: "countryName": None, "postalAddressCategory": None, "correspondentNameText": None, + "countryOrStateCode": None, + "ictStateCode": None, + "ictCountryCode": None, } assert address.to_dict() == expected_camel_case_empty_dict @@ -1237,6 +1331,8 @@ def test_assignment_from_dict(self, sample_address_data: Dict[str, Any]) -> None "assignmentRecordedDate": "2023-01-15", "assignmentMailedDate": "2023-01-20", "conveyanceText": "ASSIGNMENT OF ASSIGNORS INTEREST", + "imageAvailableStatusCode": True, + "attorneyDocketNumber": "12345-001", "assignorBag": [ {"assignorName": "John Smith", "executionDate": "2022-12-15"} ], @@ -1246,20 +1342,28 @@ def test_assignment_from_dict(self, sample_address_data: Dict[str, Any]) -> None "assigneeAddress": sample_address_data, } ], - "correspondenceAddressBag": [sample_address_data], + "correspondenceAddress": sample_address_data, + "domesticRepresentative": sample_address_data, } assignment = Assignment.from_dict(data) assert assignment.reel_number == 12345 assert assignment.frame_number == 67890 assert assignment.page_total_quantity == 3 assert assignment.assignment_received_date == date(2023, 1, 1) + assert assignment.image_available_status_code is True + assert assignment.attorney_docket_number == "12345-001" assert len(assignment.assignor_bag) == 1 assert assignment.assignor_bag[0].assignor_name == "John Smith" assert len(assignment.assignee_bag) == 1 assert assignment.assignee_bag[0].assignee_name_text == "Test Company Inc." - assert len(assignment.correspondence_address_bag) == 1 + assert assignment.correspondence_address is not None assert ( - assignment.correspondence_address_bag[0].city_name + assignment.correspondence_address.city_name + == sample_address_data["cityName"] + ) + assert assignment.domestic_representative is not None + assert ( + assignment.domestic_representative.city_name == sample_address_data["cityName"] ) @@ -1275,31 +1379,41 @@ def test_assignment_to_dict(self, sample_address_data: Dict[str, Any]) -> None: frame_number=2002, page_total_quantity=5, assignment_received_date=date(2023, 2, 1), + image_available_status_code=False, + attorney_docket_number="TEST-123", assignor_bag=[assignor_obj], assignee_bag=[assignee_obj], - correspondence_address_bag=[address_obj], + correspondence_address=address_obj, + domestic_representative=address_obj, ) data = assignment.to_dict() assert data["reelNumber"] == 1001 assert data["frameNumber"] == 2002 assert data["pageTotalQuantity"] == 5 assert data["assignmentReceivedDate"] == "2023-02-01" + assert data["imageAvailableStatusCode"] is False + assert data["attorneyDocketNumber"] == "TEST-123" assert len(data["assignorBag"]) == 1 assert len(data["assigneeBag"]) == 1 - assert len(data["correspondenceAddressBag"]) == 1 + assert data["correspondenceAddress"] is not None + assert data["correspondenceAddress"]["cityName"] == sample_address_data["cityName"] + assert data["domesticRepresentative"] is not None + assert data["domesticRepresentative"]["cityName"] == sample_address_data["cityName"] def test_assignment_to_dict_empty_bags(self) -> None: assignment = Assignment( reel_number=2002, assignor_bag=[], assignee_bag=[], - correspondence_address_bag=[], + correspondence_address=None, + domestic_representative=None, ) data = assignment.to_dict() assert data["reelNumber"] == 2002 assert data["assignorBag"] == [] assert data["assigneeBag"] == [] - assert data["correspondenceAddressBag"] == [] + assert data["correspondenceAddress"] is None + assert data["domesticRepresentative"] is None def test_assignment_roundtrip( self, @@ -1451,7 +1565,6 @@ def test_pta_history_to_dict(self) -> None: applicant_day_delay_quantity=10.0, event_description_text="Response to Office Action", event_sequence_number=1.0, - ip_office_day_delay_quantity=5.0, originating_event_sequence_number=0.0, pta_pte_code="A", ) @@ -1468,13 +1581,11 @@ def test_pta_data_from_dict(self) -> None: data = { "aDelayQuantity": 100.0, "adjustmentTotalQuantity": 150.0, - "filingDate": "2020-01-01", - "grantDate": "2023-01-01", "patentTermAdjustmentHistoryDataBag": [{"eventDate": "2022-01-01"}], } pta_data = PatentTermAdjustmentData.from_dict(data) assert pta_data.a_delay_quantity == 100.0 - assert pta_data.filing_date == date(2020, 1, 1) + assert pta_data.adjustment_total_quantity == 150.0 assert len(pta_data.patent_term_adjustment_history_data_bag) == 1 assert pta_data.patent_term_adjustment_history_data_bag[0].event_date == date( 2022, 1, 1 @@ -1484,12 +1595,12 @@ def test_pta_data_to_dict(self) -> None: history_item = PatentTermAdjustmentHistoryData(event_date=date(2022, 1, 1)) pta_data = PatentTermAdjustmentData( a_delay_quantity=100.0, - filing_date=date(2020, 1, 1), + adjustment_total_quantity=150.0, patent_term_adjustment_history_data_bag=[history_item], ) data = pta_data.to_dict() assert data["aDelayQuantity"] == 100.0 - assert data["filingDate"] == "2020-01-01" + assert data["adjustmentTotalQuantity"] == 150.0 assert len(data["patentTermAdjustmentHistoryDataBag"]) == 1 assert ( data["patentTermAdjustmentHistoryDataBag"][0]["eventDate"] == "2022-01-01"